{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/offline-rl/papers/3","list_of":"/task/offline-rl","task":"Offline RL","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":8,"rows_per_page":100,"rows":[201,300],"of":755,"counts":{"archive_papers_tagged":755,"with_a_code_link":310,"where_syntology_ran_a_sample":164,"not_listed_spam_title":0,"listed":755,"listed_where_code_ran":164,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":139,"every_run_a_failure_of_syntologys_instrument":25,"listed_with_a_run_with_no_instrument_failure":139,"listed_every_run_a_failure_of_syntologys_instrument":25,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/offline-rl","prev":"/task/offline-rl/papers/2","next":"/task/offline-rl/papers/4","papers":[{"url":"/paper/the-benefits-of-being-distributional-small-1","slug":"the-benefits-of-being-distributional-small-1","title":"The Benefits of Being Distributional: Small-Loss Bounds for Reinforcement Learning","date":"2023-05-25","arxiv_id":"2305.15703","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/the-benefits-of-being-distributional-small-1#ran","syntology_url":"https://syntology.ai/paper/2305.15703","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15703"}},"official":{"repos":["kevinzhou497/distcb"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/collaborative-world-models-an-online-offline","slug":"collaborative-world-models-an-online-offline","title":"Making Offline RL Online: Collaborative World Models for Offline Visual Reinforcement Learning","date":"2023-05-24","arxiv_id":"2305.15260","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":7,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":13,"phrase":"10 ran (of which 7 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/collaborative-world-models-an-online-offline#ran","syntology_url":"https://syntology.ai/paper/2305.15260","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15260"}},"official":{"repos":["qiwang067/CoWorld"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":7,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-language-models-with-advantage","slug":"improving-language-models-with-advantage","title":"Leftover Lunch: Advantage-based Offline Reinforcement Learning for Language Models","date":"2023-05-24","arxiv_id":"2305.14718","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/improving-language-models-with-advantage#ran","syntology_url":"https://syntology.ai/paper/2305.14718","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14718"}},"official":{"repos":["abaheti95/lol-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/2305-14550","slug":"2305-14550","title":"When should we prefer Decision Transformers for Offline Reinforcement Learning?","date":"2023-05-23","arxiv_id":"2305.14550","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/2305-14550#ran","syntology_url":"https://syntology.ai/paper/2305.14550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14550"}},"official":{"repos":["prajjwal1/rl_paradigm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/furniturebench-reproducible-real-world","slug":"furniturebench-reproducible-real-world","title":"FurnitureBench: Reproducible Real-World Benchmark for Long-Horizon Complex Manipulation","date":"2023-05-22","arxiv_id":"2305.12821","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/furniturebench-reproducible-real-world#ran","syntology_url":"https://syntology.ai/paper/2305.12821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12821"}},"official":{"repos":["clvrai/furniture-bench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/federated-ensemble-directed-offline","slug":"federated-ensemble-directed-offline","title":"Federated Ensemble-Directed Offline Reinforcement Learning","date":"2023-05-04","arxiv_id":"2305.03097","repositories_listed":1,"syntology":null},{"url":"/paper/masked-trajectory-models-for-prediction","slug":"masked-trajectory-models-for-prediction","title":"Masked Trajectory Models for Prediction, Representation, and Control","date":"2023-05-04","arxiv_id":"2305.02968","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/masked-trajectory-models-for-prediction#ran","syntology_url":"https://syntology.ai/paper/2305.02968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.02968"}},"official":{"repos":["facebookresearch/mtm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/idql-implicit-q-learning-as-an-actor-critic","slug":"idql-implicit-q-learning-as-an-actor-critic","title":"IDQL: Implicit Q-Learning as an Actor-Critic Method with Diffusion Policies","date":"2023-04-20","arxiv_id":"2304.10573","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/idql-implicit-q-learning-as-an-actor-critic#ran","syntology_url":"https://syntology.ai/paper/2304.10573","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.10573"}},"official":{"repos":["philippe-eecs/idql"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/using-offline-data-to-speed-up-reinforcement","slug":"using-offline-data-to-speed-up-reinforcement","title":"Using Offline Data to Speed Up Reinforcement Learning in Procedurally Generated Environments","date":"2023-04-18","arxiv_id":"2304.09825","repositories_listed":1,"syntology":null},{"url":"/paper/uncertainty-driven-trajectory-truncation-for","slug":"uncertainty-driven-trajectory-truncation-for","title":"Uncertainty-driven Trajectory Truncation for Data Augmentation in Offline Reinforcement Learning","date":"2023-04-10","arxiv_id":"2304.04660","repositories_listed":1,"syntology":null},{"url":"/paper/mahalo-unifying-offline-reinforcement","slug":"mahalo-unifying-offline-reinforcement","title":"MAHALO: Unifying Offline Reinforcement Learning and Imitation Learning from Observations","date":"2023-03-30","arxiv_id":"2303.17156","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mahalo-unifying-offline-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2303.17156","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17156"}},"official":{"repos":["anqili/mahalo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/optimal-transport-for-offline-imitation","slug":"optimal-transport-for-offline-imitation","title":"Optimal Transport for Offline Imitation Learning","date":"2023-03-24","arxiv_id":"2303.13971","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimal-transport-for-offline-imitation#ran","syntology_url":"https://syntology.ai/paper/2303.13971","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.13971"}},"official":{"repos":["ethanluoyc/optimal_transport_reward"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/data-might-be-enough-bridge-real-world","slug":"data-might-be-enough-bridge-real-world","title":"DataLight: Offline Data-Driven Traffic Signal Control","date":"2023-03-20","arxiv_id":"2303.10828","repositories_listed":1,"syntology":null},{"url":"/paper/decision-transformer-under-random-frame","slug":"decision-transformer-under-random-frame","title":"Decision Transformer under Random Frame Dropping","date":"2023-03-03","arxiv_id":"2303.03391","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":3,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/decision-transformer-under-random-frame#ran","syntology_url":"https://syntology.ai/paper/2303.03391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03391"}},"official":{"repos":["hukz18/defog"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/provably-efficient-neural-offline","slug":"provably-efficient-neural-offline","title":"VIPeR: Provably Efficient Algorithm for Offline RL with Neural Function Approximation","date":"2023-02-24","arxiv_id":"2302.12780","repositories_listed":1,"syntology":null},{"url":"/paper/swapped-goal-conditioned-offline","slug":"swapped-goal-conditioned-offline","title":"Swapped goal-conditioned offline reinforcement learning","date":"2023-02-17","arxiv_id":"2302.08865","repositories_listed":1,"syntology":null},{"url":"/paper/imitation-from-arbitrary-experience-a-dual","slug":"imitation-from-arbitrary-experience-a-dual","title":"Dual RL: Unification and New Methods for Reinforcement and Imitation Learning","date":"2023-02-16","arxiv_id":"2302.08560","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imitation-from-arbitrary-experience-a-dual#ran","syntology_url":"https://syntology.ai/paper/2302.08560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.08560"}},"official":{"repos":["hari-sikchi/DVL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-bellman-errors-for-offline-model","slug":"revisiting-bellman-errors-for-offline-model","title":"Revisiting Bellman Errors for Offline Model Selection","date":"2023-01-31","arxiv_id":"2302.00141","repositories_listed":1,"syntology":null},{"url":"/paper/guiding-online-reinforcement-learning-with","slug":"guiding-online-reinforcement-learning-with","title":"Guiding Online Reinforcement Learning with Action-Free Offline Pretraining","date":"2023-01-30","arxiv_id":"2301.12876","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/guiding-online-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2301.12876","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12876"}},"official":{"repos":["vision-cair/af-guide"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/winning-solution-of-real-robot-challenge-iii","slug":"winning-solution-of-real-robot-challenge-iii","title":"Identifying Expert Behavior in Offline Training Datasets Improves Behavioral Cloning of Robotic Manipulation Policies","date":"2023-01-30","arxiv_id":"2301.13019","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/winning-solution-of-real-robot-challenge-iii#ran","syntology_url":"https://syntology.ai/paper/2301.13019","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.13019"}},"official":{"repos":["wq13552463699/real-robot-challenge-2022"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-reinforcement-learning-for-visual","slug":"offline-reinforcement-learning-for-visual","title":"Offline Reinforcement Learning for Visual Navigation","date":"2022-12-16","arxiv_id":"2212.08244","repositories_listed":1,"syntology":null},{"url":"/paper/td3-with-reverse-kl-regularizer-for-offline","slug":"td3-with-reverse-kl-regularizer-for-offline","title":"TD3 with Reverse KL Regularizer for Offline Reinforcement Learning from Mixed Datasets","date":"2022-12-05","arxiv_id":"2212.02125","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-reinforcement-learning-through","slug":"efficient-reinforcement-learning-through","title":"Efficient Reinforcement Learning Through Trajectory Generation","date":"2022-11-30","arxiv_id":"2211.17249","repositories_listed":1,"syntology":null},{"url":"/paper/one-risk-to-rule-them-all-a-risk-sensitive-1","slug":"one-risk-to-rule-them-all-a-risk-sensitive-1","title":"One Risk to Rule Them All: A Risk-Sensitive Perspective on Model-Based Offline Reinforcement Learning","date":"2022-11-30","arxiv_id":"2212.00124","repositories_listed":1,"syntology":{"n":23,"n_ran":14,"n_constructed":4,"n_ran_checked":8,"n_instrument":6,"n_unverified":9,"n_honours":2,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"14 ran (of which 4 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 0 violated, 6 with no contract checked; 6 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/one-risk-to-rule-them-all-a-risk-sensitive-1#ran","syntology_url":"https://syntology.ai/paper/2212.00124","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.00124"}},"official":{"repos":["marc-rigter/1r2r"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["found_in_text","official","unlocated"]}}},{"url":"/paper/behavior-estimation-from-multi-source-data","slug":"behavior-estimation-from-multi-source-data","title":"Behavior Estimation from Multi-Source Data for Offline Reinforcement Learning","date":"2022-11-29","arxiv_id":"2211.16078","repositories_listed":1,"syntology":null},{"url":"/paper/masked-autoencoding-for-scalable-and","slug":"masked-autoencoding-for-scalable-and","title":"Masked Autoencoding for Scalable and Generalizable Decision Making","date":"2022-11-23","arxiv_id":"2211.12740","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/masked-autoencoding-for-scalable-and#ran","syntology_url":"https://syntology.ai/paper/2211.12740","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.12740"}},"official":{"repos":["fangchenliu/maskdp_public"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-low-latency-adaptive-coding-spiking","slug":"a-low-latency-adaptive-coding-spiking","title":"A Low Latency Adaptive Coding Spiking Framework for Deep Reinforcement Learning","date":"2022-11-21","arxiv_id":"2211.11760","repositories_listed":1,"syntology":null},{"url":"/paper/behavior-prior-representation-learning-for","slug":"behavior-prior-representation-learning-for","title":"Behavior Prior Representation learning for Offline Reinforcement Learning","date":"2022-11-02","arxiv_id":"2211.00863","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/behavior-prior-representation-learning-for#ran","syntology_url":"https://syntology.ai/paper/2211.00863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.00863"}},"official":{"repos":["bit1029public/offline_bpr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dungeons-and-data-a-large-scale-nethack","slug":"dungeons-and-data-a-large-scale-nethack","title":"Dungeons and Data: A Large-Scale NetHack Dataset","date":"2022-11-01","arxiv_id":"2211.00539","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dungeons-and-data-a-large-scale-nethack#ran","syntology_url":"https://syntology.ai/paper/2211.00539","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.00539"}},"official":{"repos":["facebookresearch/nle"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-demonstrations-with-latent-space","slug":"leveraging-demonstrations-with-latent-space","title":"Leveraging Demonstrations with Latent Space Priors","date":"2022-10-26","arxiv_id":"2210.14685","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leveraging-demonstrations-with-latent-space#ran","syntology_url":"https://syntology.ai/paper/2210.14685","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.14685"}},"official":{"repos":["facebookresearch/latent-space-priors"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mocoda-model-based-counterfactual-data","slug":"mocoda-model-based-counterfactual-data","title":"MoCoDA: Model-based Counterfactual Data Augmentation","date":"2022-10-20","arxiv_id":"2210.11287","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mocoda-model-based-counterfactual-data#ran","syntology_url":"https://syntology.ai/paper/2210.11287","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.11287"}},"official":{"repos":["spitis/mocoda"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/the-pump-scheduling-problem-a-real-world","slug":"the-pump-scheduling-problem-a-real-world","title":"The Pump Scheduling Problem: A Real-World Scenario for Reinforcement Learning","date":"2022-10-20","arxiv_id":"2210.11111","repositories_listed":1,"syntology":null},{"url":"/paper/robust-offline-reinforcement-learning-with","slug":"robust-offline-reinforcement-learning-with","title":"Robust Offline Reinforcement Learning with Gradient Penalty and Constraint Relaxation","date":"2022-10-19","arxiv_id":"2210.10469","repositories_listed":1,"syntology":null},{"url":"/paper/a-policy-guided-imitation-approach-for","slug":"a-policy-guided-imitation-approach-for","title":"A Policy-Guided Imitation Approach for Offline Reinforcement Learning","date":"2022-10-15","arxiv_id":"2210.08323","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/a-policy-guided-imitation-approach-for#ran","syntology_url":"https://syntology.ai/paper/2210.08323","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.08323"}},"official":{"repos":["ryanxhr/por"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mutual-information-regularized-offline-1","slug":"mutual-information-regularized-offline-1","title":"Mutual Information Regularized Offline Reinforcement Learning","date":"2022-10-14","arxiv_id":"2210.07484","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mutual-information-regularized-offline-1#ran","syntology_url":"https://syntology.ai/paper/2210.07484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07484"}},"official":{"repos":["sail-sg/misa"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-offline-policy-optimization-with-a","slug":"efficient-offline-policy-optimization-with-a","title":"Efficient Offline Policy Optimization with a Learned Model","date":"2022-10-12","arxiv_id":"2210.05980","repositories_listed":1,"syntology":null},{"url":"/paper/semi-supervised-offline-reinforcement-1","slug":"semi-supervised-offline-reinforcement-1","title":"Semi-Supervised Offline Reinforcement Learning with Action-Free Trajectories","date":"2022-10-12","arxiv_id":"2210.06518","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/semi-supervised-offline-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2210.06518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06518"}},"official":{"repos":["facebookresearch/ssorl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/conserweightive-behavioral-cloning-for","slug":"conserweightive-behavioral-cloning-for","title":"Reliable Conditioning of Behavioral Cloning for Offline Reinforcement Learning","date":"2022-10-11","arxiv_id":"2210.05158","repositories_listed":1,"syntology":null},{"url":"/paper/pre-training-for-robots-offline-rl-enables","slug":"pre-training-for-robots-offline-rl-enables","title":"Pre-Training for Robots: Offline RL Enables Learning New Tasks from a Handful of Trials","date":"2022-10-11","arxiv_id":"2210.05178","repositories_listed":1,"syntology":null},{"url":"/paper/mind-your-data-hiding-backdoors-in-offline","slug":"mind-your-data-hiding-backdoors-in-offline","title":"BAFFLE: Hiding Backdoors in Offline Reinforcement Learning Datasets","date":"2022-10-07","arxiv_id":"2210.04688","repositories_listed":1,"syntology":null},{"url":"/paper/s2p-state-conditioned-image-synthesis-for","slug":"s2p-state-conditioned-image-synthesis-for","title":"S2P: State-conditioned Image Synthesis for Data Augmentation in Offline Reinforcement Learning","date":"2022-09-30","arxiv_id":"2209.15256","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/s2p-state-conditioned-image-synthesis-for#ran","syntology_url":"https://syntology.ai/paper/2209.15256","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.15256"}},"official":{"repos":["dsshim0125/s2p"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vip-towards-universal-visual-reward-and","slug":"vip-towards-universal-visual-reward-and","title":"VIP: Towards Universal Visual Reward and Representation via Value-Implicit Pre-Training","date":"2022-09-30","arxiv_id":"2210.00030","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vip-towards-universal-visual-reward-and#ran","syntology_url":"https://syntology.ai/paper/2210.00030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.00030"}},"official":{"repos":["facebookresearch/vip"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-reinforcement-learning-via-high","slug":"offline-reinforcement-learning-via-high","title":"Offline Reinforcement Learning via High-Fidelity Generative Behavior Modeling","date":"2022-09-29","arxiv_id":"2209.14548","repositories_listed":1,"syntology":null},{"url":"/paper/exploiting-reward-shifting-in-value-based","slug":"exploiting-reward-shifting-in-value-based","title":"Optimistic Curiosity Exploration and Conservative Exploitation with Linear Reward Shaping","date":"2022-09-15","arxiv_id":"2209.07288","repositories_listed":1,"syntology":null},{"url":"/paper/q-learning-decision-transformer-leveraging","slug":"q-learning-decision-transformer-leveraging","title":"Q-learning Decision Transformer: Leveraging Dynamic Programming for Conditional Sequence Modelling in Offline RL","date":"2022-09-08","arxiv_id":"2209.03993","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-planning-in-a-compact-latent-action","slug":"efficient-planning-in-a-compact-latent-action","title":"Efficient Planning in a Compact Latent Action Space","date":"2022-08-22","arxiv_id":"2208.10291","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-planning-in-a-compact-latent-action#ran","syntology_url":"https://syntology.ai/paper/2208.10291","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.10291"}},"official":{"repos":["ZhengyaoJiang/latentplan"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/adacat-adaptive-categorical-discretization","slug":"adacat-adaptive-categorical-discretization","title":"AdaCat: Adaptive Categorical Discretization for Autoregressive Models","date":"2022-08-03","arxiv_id":"2208.02246","repositories_listed":1,"syntology":null},{"url":"/paper/offline-equilibrium-finding","slug":"offline-equilibrium-finding","title":"Offline Equilibrium Finding","date":"2022-07-12","arxiv_id":"2207.05285","repositories_listed":1,"syntology":null},{"url":"/paper/when-to-trust-your-simulator-dynamics-aware","slug":"when-to-trust-your-simulator-dynamics-aware","title":"When to Trust Your Simulator: Dynamics-Aware Hybrid Offline-and-Online Reinforcement Learning","date":"2022-06-27","arxiv_id":"2206.13464","repositories_listed":1,"syntology":null},{"url":"/paper/towards-human-level-bimanual-dexterous","slug":"towards-human-level-bimanual-dexterous","title":"Towards Human-Level Bimanual Dexterous Manipulation with Reinforcement Learning","date":"2022-06-17","arxiv_id":"2206.08686","repositories_listed":1,"syntology":null},{"url":"/paper/double-check-your-state-before-trusting-it","slug":"double-check-your-state-before-trusting-it","title":"Double Check Your State Before Trusting It: Confidence-Aware Bidirectional Offline Model-Based Imagination","date":"2022-06-16","arxiv_id":"2206.07989","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":1,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/double-check-your-state-before-trusting-it#ran","syntology_url":"https://syntology.ai/paper/2206.07989","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07989"}},"official":{"repos":["dmksjfl/CABI"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/regularizing-a-model-based-policy-stationary","slug":"regularizing-a-model-based-policy-stationary","title":"Regularizing a Model-based Policy Stationary Distribution to Stabilize Offline Reinforcement Learning","date":"2022-06-14","arxiv_id":"2206.07166","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regularizing-a-model-based-policy-stationary#ran","syntology_url":"https://syntology.ai/paper/2206.07166","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07166"}},"official":{"repos":["shentao-yang/sdm-gan_icml2022"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/value-memory-graph-a-graph-structured-world","slug":"value-memory-graph-a-graph-structured-world","title":"Value Memory Graph: A Graph-Structured World Model for Offline Reinforcement Learning","date":"2022-06-09","arxiv_id":"2206.04384","repositories_listed":1,"syntology":null},{"url":"/paper/rorl-robust-offline-reinforcement-learning","slug":"rorl-robust-offline-reinforcement-learning","title":"RORL: Robust Offline Reinforcement Learning via Conservative Smoothing","date":"2022-06-06","arxiv_id":"2206.02829","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rorl-robust-offline-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2206.02829","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.02829"}},"official":{"repos":["yangrui2015/rorl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-game-decision-transformers","slug":"multi-game-decision-transformers","title":"Multi-Game Decision Transformers","date":"2022-05-30","arxiv_id":"2205.15241","repositories_listed":1,"syntology":null},{"url":"/paper/user-interactive-offline-reinforcement","slug":"user-interactive-offline-reinforcement","title":"User-Interactive Offline Reinforcement Learning","date":"2022-05-21","arxiv_id":"2205.10629","repositories_listed":1,"syntology":null},{"url":"/paper/coptidice-offline-constrained-reinforcement-1","slug":"coptidice-offline-constrained-reinforcement-1","title":"COptiDICE: Offline Constrained Reinforcement Learning via Stationary Distribution Correction Estimation","date":"2022-04-19","arxiv_id":"2204.08957","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/coptidice-offline-constrained-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2204.08957","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.08957"}},"official":{"repos":["deepmind/constrained_optidice"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-reinforcement-learning-for-safer","slug":"offline-reinforcement-learning-for-safer","title":"Offline Reinforcement Learning for Safer Blood Glucose Control in People with Type 1 Diabetes","date":"2022-04-07","arxiv_id":"2204.03376","repositories_listed":1,"syntology":null},{"url":"/paper/cirs-bursting-filter-bubbles-by","slug":"cirs-bursting-filter-bubbles-by","title":"CIRS: Bursting Filter Bubbles by Counterfactual Interactive Recommender System","date":"2022-04-04","arxiv_id":"2204.01266","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cirs-bursting-filter-bubbles-by#ran","syntology_url":"https://syntology.ai/paper/2204.01266","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.01266"}},"official":{"repos":["chongminggao/cirs-codes"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/semi-markov-offline-reinforcement-learning","slug":"semi-markov-offline-reinforcement-learning","title":"Semi-Markov Offline Reinforcement Learning for Healthcare","date":"2022-03-17","arxiv_id":"2203.09365","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/semi-markov-offline-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2203.09365","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.09365"}},"official":{"repos":["mary-wu/smdp"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/copa-certifying-robust-policies-for-offline-1","slug":"copa-certifying-robust-policies-for-offline-1","title":"COPA: Certifying Robust Policies for Offline Reinforcement Learning against Poisoning Attacks","date":"2022-03-16","arxiv_id":"2203.08398","repositories_listed":1,"syntology":null},{"url":"/paper/latent-variable-advantage-weighted-policy","slug":"latent-variable-advantage-weighted-policy","title":"Latent-Variable Advantage-Weighted Policy Optimization for Offline RL","date":"2022-03-16","arxiv_id":"2203.08949","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/latent-variable-advantage-weighted-policy#ran","syntology_url":"https://syntology.ai/paper/2203.08949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08949"}},"official":null}},{"url":"/paper/on-practical-reinforcement-learning-provable","slug":"on-practical-reinforcement-learning-provable","title":"On Practical Reinforcement Learning: Provable Robustness, Scalability, and Statistical Efficiency","date":"2022-03-03","arxiv_id":"2203.01758","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-offline-reinforcement-learning","slug":"a-survey-on-offline-reinforcement-learning","title":"A Survey on Offline Reinforcement Learning: Taxonomy, Review, and Open Problems","date":"2022-03-02","arxiv_id":"2203.01387","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-survey-on-offline-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2203.01387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.01387"}},"official":{"repos":["larocs/offline-rl-suvey"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/all-you-need-is-supervised-learning-from","slug":"all-you-need-is-supervised-learning-from","title":"All You Need Is Supervised Learning: From Imitation Learning to Meta-RL With Upside Down RL","date":"2022-02-24","arxiv_id":"2202.11960","repositories_listed":1,"syntology":null},{"url":"/paper/pessimistic-bootstrapping-for-uncertainty-1","slug":"pessimistic-bootstrapping-for-uncertainty-1","title":"Pessimistic Bootstrapping for Uncertainty-Driven Offline Reinforcement Learning","date":"2022-02-23","arxiv_id":"2202.11566","repositories_listed":1,"syntology":null},{"url":"/paper/vrl3-a-data-driven-framework-for-visual-deep","slug":"vrl3-a-data-driven-framework-for-visual-deep","title":"VRL3: A Data-Driven Framework for Visual Deep Reinforcement Learning","date":"2022-02-17","arxiv_id":"2202.10324","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":4,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/vrl3-a-data-driven-framework-for-visual-deep#ran","syntology_url":"https://syntology.ai/paper/2202.10324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.10324"}},"official":{"repos":["facebookresearch/drqv2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/flowformer-linearizing-transformers-with","slug":"flowformer-linearizing-transformers-with","title":"Flowformer: Linearizing Transformers with Conservation Flows","date":"2022-02-13","arxiv_id":"2202.06258","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/flowformer-linearizing-transformers-with#ran","syntology_url":"https://syntology.ai/paper/2202.06258","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.06258"}},"official":{"repos":["thuml/Flowformer"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-goal-conditioned-supervised-1","slug":"rethinking-goal-conditioned-supervised-1","title":"Rethinking Goal-conditioned Supervised Learning and Its Connection to Offline RL","date":"2022-02-09","arxiv_id":"2202.04478","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rethinking-goal-conditioned-supervised-1#ran","syntology_url":"https://syntology.ai/paper/2202.04478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.04478"}},"official":{"repos":["yangrui2015/awgcsl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/don-t-change-the-algorithm-change-the-data","slug":"don-t-change-the-algorithm-change-the-data","title":"Don't Change the Algorithm, Change the Data: Exploratory Data for Offline Reinforcement Learning","date":"2022-01-31","arxiv_id":"2201.13425","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/don-t-change-the-algorithm-change-the-data#ran","syntology_url":"https://syntology.ai/paper/2201.13425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.13425"}},"official":{"repos":["denisyarats/exorl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-wikipedia-help-offline-reinforcement","slug":"can-wikipedia-help-offline-reinforcement","title":"Can Wikipedia Help Offline Reinforcement Learning?","date":"2022-01-28","arxiv_id":"2201.12122","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-wikipedia-help-offline-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2201.12122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12122"}},"official":{"repos":["machelreid/can-wikipedia-help-offline-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/comparing-model-free-and-model-based","slug":"comparing-model-free-and-model-based","title":"Comparing Model-free and Model-based Algorithms for Offline Reinforcement Learning","date":"2022-01-14","arxiv_id":"2201.05433","repositories_listed":1,"syntology":null},{"url":"/paper/rvs-what-is-essential-for-offline-rl-via","slug":"rvs-what-is-essential-for-offline-rl-via","title":"RvS: What is Essential for Offline RL via Supervised Learning?","date":"2021-12-20","arxiv_id":"2112.10751","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rvs-what-is-essential-for-offline-rl-via#ran","syntology_url":"https://syntology.ai/paper/2112.10751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.10751"}},"official":{"repos":["scottemmons/rvs"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-pre-trained-multi-agent-decision-1","slug":"offline-pre-trained-multi-agent-decision-1","title":"Offline Pre-trained Multi-Agent Decision Transformer: One Big Sequence Model Tackles All SMAC Tasks","date":"2021-12-06","arxiv_id":"2112.02845","repositories_listed":1,"syntology":null},{"url":"/paper/robust-on-policy-data-collection-for-data","slug":"robust-on-policy-data-collection-for-data","title":"Robust On-Policy Sampling for Data-Efficient Policy Evaluation in Reinforcement Learning","date":"2021-11-29","arxiv_id":"2111.14552","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-on-policy-data-collection-for-data#ran","syntology_url":"https://syntology.ai/paper/2111.14552","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.14552"}},"official":{"repos":["uoe-agents/robust_onpolicy_data_collection"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/plan-better-amid-conservatism-offline-multi-1","slug":"plan-better-amid-conservatism-offline-multi-1","title":"Plan Better Amid Conservatism: Offline Multi-Agent Reinforcement Learning with Actor Rectification","date":"2021-11-22","arxiv_id":"2111.11188","repositories_listed":1,"syntology":null},{"url":"/paper/rlds-an-ecosystem-to-generate-share-and-use","slug":"rlds-an-ecosystem-to-generate-share-and-use","title":"RLDS: an Ecosystem to Generate, Share and Use Datasets in Reinforcement Learning","date":"2021-11-04","arxiv_id":"2111.02767","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rlds-an-ecosystem-to-generate-share-and-use#ran","syntology_url":"https://syntology.ai/paper/2111.02767","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.02767"}},"official":{"repos":["google-research/rlds"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/curriculum-offline-imitation-learning","slug":"curriculum-offline-imitation-learning","title":"Curriculum Offline Imitation Learning","date":"2021-11-03","arxiv_id":"2111.02056","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/curriculum-offline-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/2111.02056","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.02056"}},"official":{"repos":["apexrl/coil"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/score-spurious-correlation-reduction-for","slug":"score-spurious-correlation-reduction-for","title":"False Correlation Reduction for Offline Reinforcement Learning","date":"2021-10-24","arxiv_id":"2110.12468","repositories_listed":1,"syntology":null},{"url":"/paper/offline-reinforcement-learning-with-value-1","slug":"offline-reinforcement-learning-with-value-1","title":"Offline Reinforcement Learning with Value-based Episodic Memory","date":"2021-10-19","arxiv_id":"2110.09796","repositories_listed":1,"syntology":null},{"url":"/paper/safe-driving-via-expert-guided-policy","slug":"safe-driving-via-expert-guided-policy","title":"Safe Driving via Expert Guided Policy Optimization","date":"2021-10-13","arxiv_id":"2110.06831","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safe-driving-via-expert-guided-policy#ran","syntology_url":"https://syntology.ai/paper/2110.06831","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06831"}},"official":{"repos":["decisionforce/EGPO"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-pick-and-place-tackling-robotic","slug":"beyond-pick-and-place-tackling-robotic","title":"Beyond Pick-and-Place: Tackling Robotic Stacking of Diverse Shapes","date":"2021-10-12","arxiv_id":"2110.06192","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-pick-and-place-tackling-robotic#ran","syntology_url":"https://syntology.ai/paper/2110.06192","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06192"}},"official":{"repos":["deepmind/rgb_stacking"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/starformer-transformer-with-state-action-1","slug":"starformer-transformer-with-state-action-1","title":"StARformer: Transformer with State-Action-Reward Representations for Visual Reinforcement Learning","date":"2021-10-12","arxiv_id":"2110.06206","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/starformer-transformer-with-state-action-1#ran","syntology_url":"https://syntology.ai/paper/2110.06206","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06206"}},"official":{"repos":["elicassion/StARformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/brac-improved-behavior-regularized-actor","slug":"brac-improved-behavior-regularized-actor","title":"BRAC+: Improved Behavior Regularized Actor Critic for Offline Reinforcement Learning","date":"2021-10-02","arxiv_id":"2110.00894","repositories_listed":1,"syntology":null},{"url":"/paper/offline-reinforcement-learning-with-reverse","slug":"offline-reinforcement-learning-with-reverse","title":"Offline Reinforcement Learning with Reverse Model-based Imagination","date":"2021-10-01","arxiv_id":"2110.00188","repositories_listed":1,"syntology":null},{"url":"/paper/offline-reinforcement-learning-with-in-sample","slug":"offline-reinforcement-learning-with-in-sample","title":"Offline Reinforcement Learning with In-sample Q-Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-workflow-for-offline-model-free-robotic","slug":"a-workflow-for-offline-model-free-robotic","title":"A Workflow for Offline Model-Free Robotic Reinforcement Learning","date":"2021-09-22","arxiv_id":"2109.10813","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-workflow-for-offline-model-free-robotic#ran","syntology_url":"https://syntology.ai/paper/2109.10813","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.10813"}},"official":null}},{"url":"/paper/dcur-data-curriculum-for-teaching-via-samples","slug":"dcur-data-curriculum-for-teaching-via-samples","title":"DCUR: Data Curriculum for Teaching via Samples with Reinforcement Learning","date":"2021-09-15","arxiv_id":"2109.07380","repositories_listed":1,"syntology":null},{"url":"/paper/model-selection-for-offline-reinforcement","slug":"model-selection-for-offline-reinforcement","title":"Model Selection for Offline Reinforcement Learning: Practical Considerations for Healthcare Settings","date":"2021-07-23","arxiv_id":"2107.11003","repositories_listed":1,"syntology":null},{"url":"/paper/conservative-offline-distributional","slug":"conservative-offline-distributional","title":"Conservative Offline Distributional Reinforcement Learning","date":"2021-07-12","arxiv_id":"2107.06106","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/conservative-offline-distributional#ran","syntology_url":"https://syntology.ai/paper/2107.06106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.06106"}},"official":{"repos":["JasonMa2016/CODAC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-meta-reinforcement-learning-with-1","slug":"offline-meta-reinforcement-learning-with-1","title":"Offline Meta-Reinforcement Learning with Online Self-Supervision","date":"2021-07-08","arxiv_id":"2107.03974","repositories_listed":1,"syntology":null},{"url":"/paper/where-is-the-grass-greener-revisiting","slug":"where-is-the-grass-greener-revisiting","title":"Optimality Inductive Biases and Agnostic Guidelines for Offline Reinforcement Learning","date":"2021-07-03","arxiv_id":"2107.01407","repositories_listed":1,"syntology":null},{"url":"/paper/offline-to-online-reinforcement-learning-via","slug":"offline-to-online-reinforcement-learning-via","title":"Offline-to-Online Reinforcement Learning via Balanced Replay and Pessimistic Q-Ensemble","date":"2021-07-01","arxiv_id":"2107.00591","repositories_listed":1,"syntology":null},{"url":"/paper/optidice-offline-policy-optimization-via","slug":"optidice-offline-policy-optimization-via","title":"OptiDICE: Offline Policy Optimization via Stationary Distribution Correction Estimation","date":"2021-06-21","arxiv_id":"2106.10783","repositories_listed":1,"syntology":null},{"url":"/paper/offline-rl-without-off-policy-evaluation","slug":"offline-rl-without-off-policy-evaluation","title":"Offline RL Without Off-Policy Evaluation","date":"2021-06-16","arxiv_id":"2106.08909","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-as-one-big-sequence-1","slug":"reinforcement-learning-as-one-big-sequence-1","title":"Reinforcement Learning as One Big Sequence Modeling Problem","date":"2021-06-13","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/believe-what-you-see-implicit-constraint","slug":"believe-what-you-see-implicit-constraint","title":"Believe What You See: Implicit Constraint Approach for Offline Multi-Agent Reinforcement Learning","date":"2021-06-07","arxiv_id":"2106.03400","repositories_listed":1,"syntology":null},{"url":"/paper/online-reinforcement-learning-with-sparse","slug":"online-reinforcement-learning-with-sparse","title":"Online reinforcement learning with sparse rewards through an active inference capsule","date":"2021-06-04","arxiv_id":"2106.02390","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-offline-planning-with-trajectory","slug":"model-based-offline-planning-with-trajectory","title":"Model-Based Offline Planning with Trajectory Pruning","date":"2021-05-16","arxiv_id":"2105.07351","repositories_listed":1,"syntology":null},{"url":"/paper/model-free-two-step-design-for-improving","slug":"model-free-two-step-design-for-improving","title":"Two-step reinforcement learning for model-free redesign of nonlinear optimal regulator","date":"2021-03-05","arxiv_id":"2103.03808","repositories_listed":1,"syntology":null}],"record_sha256":"4a47daf7fbc7a3f774ccd3fff1348485727c6f46e501947759d7d3043f9ce5cd","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}