{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/offline-rl/papers/ran/2","list_of":"/task/offline-rl","task":"Offline RL","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,164],"of":164,"counts":{"archive_papers_tagged":755,"with_a_code_link":310,"where_syntology_ran_a_sample":164,"not_listed_spam_title":0,"listed":755,"listed_where_code_ran":164,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":139,"every_run_a_failure_of_syntologys_instrument":25,"listed_with_a_run_with_no_instrument_failure":139,"listed_every_run_a_failure_of_syntologys_instrument":25,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/offline-rl/papers/ran/1","prev":"/task/offline-rl/papers/ran/1","next":null,"papers":[{"url":"/paper/winning-solution-of-real-robot-challenge-iii","slug":"winning-solution-of-real-robot-challenge-iii","title":"Identifying Expert Behavior in Offline Training Datasets Improves Behavioral Cloning of Robotic Manipulation Policies","date":"2023-01-30","arxiv_id":"2301.13019","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/winning-solution-of-real-robot-challenge-iii#ran","syntology_url":"https://syntology.ai/paper/2301.13019","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.13019"}},"official":{"repos":["wq13552463699/real-robot-challenge-2022"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/extreme-q-learning-maxent-rl-without-entropy","slug":"extreme-q-learning-maxent-rl-without-entropy","title":"Extreme Q-Learning: MaxEnt RL without Entropy","date":"2023-01-05","arxiv_id":"2301.02328","repositories_listed":4,"syntology":{"n":13,"n_ran":8,"n_constructed":5,"n_ran_checked":5,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"8 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/extreme-q-learning-maxent-rl-without-entropy#ran","syntology_url":"https://syntology.ai/paper/2301.02328","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.02328"}},"official":null}},{"url":"/paper/one-risk-to-rule-them-all-a-risk-sensitive-1","slug":"one-risk-to-rule-them-all-a-risk-sensitive-1","title":"One Risk to Rule Them All: A Risk-Sensitive Perspective on Model-Based Offline Reinforcement Learning","date":"2022-11-30","arxiv_id":"2212.00124","repositories_listed":1,"syntology":{"n":23,"n_ran":14,"n_constructed":4,"n_ran_checked":8,"n_instrument":6,"n_unverified":9,"n_honours":2,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"14 ran (of which 4 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 0 violated, 6 with no contract checked; 6 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/one-risk-to-rule-them-all-a-risk-sensitive-1#ran","syntology_url":"https://syntology.ai/paper/2212.00124","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.00124"}},"official":{"repos":["marc-rigter/1r2r"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["found_in_text","official","unlocated"]}}},{"url":"/paper/masked-autoencoding-for-scalable-and","slug":"masked-autoencoding-for-scalable-and","title":"Masked Autoencoding for Scalable and Generalizable Decision Making","date":"2022-11-23","arxiv_id":"2211.12740","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/masked-autoencoding-for-scalable-and#ran","syntology_url":"https://syntology.ai/paper/2211.12740","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.12740"}},"official":{"repos":["fangchenliu/maskdp_public"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/q-ensemble-for-offline-rl-don-t-scale-the","slug":"q-ensemble-for-offline-rl-don-t-scale-the","title":"Q-Ensemble for Offline RL: Don't Scale the Ensemble, Scale the Batch Size","date":"2022-11-20","arxiv_id":"2211.11092","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/q-ensemble-for-offline-rl-don-t-scale-the#ran","syntology_url":"https://syntology.ai/paper/2211.11092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.11092"}},"official":{"repos":["corl-team/CORL"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["named_in_paper"]}}},{"url":"/paper/let-offline-rl-flow-training-conservative","slug":"let-offline-rl-flow-training-conservative","title":"Let Offline RL Flow: Training Conservative Agents in the Latent Space of Normalizing Flows","date":"2022-11-20","arxiv_id":"2211.11096","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/let-offline-rl-flow-training-conservative#ran","syntology_url":"https://syntology.ai/paper/2211.11096","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.11096"}},"official":{"repos":["tinkoff-ai/cnf"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/behavior-prior-representation-learning-for","slug":"behavior-prior-representation-learning-for","title":"Behavior Prior Representation learning for Offline Reinforcement Learning","date":"2022-11-02","arxiv_id":"2211.00863","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/behavior-prior-representation-learning-for#ran","syntology_url":"https://syntology.ai/paper/2211.00863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.00863"}},"official":{"repos":["bit1029public/offline_bpr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dual-generator-offline-reinforcement-learning","slug":"dual-generator-offline-reinforcement-learning","title":"Dual Generator Offline Reinforcement Learning","date":"2022-11-02","arxiv_id":"2211.01471","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dual-generator-offline-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2211.01471","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.01471"}},"official":null}},{"url":"/paper/dungeons-and-data-a-large-scale-nethack","slug":"dungeons-and-data-a-large-scale-nethack","title":"Dungeons and Data: A Large-Scale NetHack Dataset","date":"2022-11-01","arxiv_id":"2211.00539","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dungeons-and-data-a-large-scale-nethack#ran","syntology_url":"https://syntology.ai/paper/2211.00539","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.00539"}},"official":{"repos":["facebookresearch/nle"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/agent-controller-representations-principled","slug":"agent-controller-representations-principled","title":"Agent-Controller Representations: Principled Offline RL with Rich Exogenous Information","date":"2022-10-31","arxiv_id":"2211.00164","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":6,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 6 samples that ran constructed an object rather than computing a result","sample_list":"/paper/agent-controller-representations-principled#ran","syntology_url":"https://syntology.ai/paper/2211.00164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.00164"}},"official":{"repos":["manantomar/agent-centric-representations"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":6,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-demonstrations-with-latent-space","slug":"leveraging-demonstrations-with-latent-space","title":"Leveraging Demonstrations with Latent Space Priors","date":"2022-10-26","arxiv_id":"2210.14685","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leveraging-demonstrations-with-latent-space#ran","syntology_url":"https://syntology.ai/paper/2210.14685","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.14685"}},"official":{"repos":["facebookresearch/latent-space-priors"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-behavior-cloning-regularization-for-1","slug":"adaptive-behavior-cloning-regularization-for-1","title":"Adaptive Behavior Cloning Regularization for Stable Offline-to-Online Reinforcement Learning","date":"2022-10-25","arxiv_id":"2210.13846","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptive-behavior-cloning-regularization-for-1#ran","syntology_url":"https://syntology.ai/paper/2210.13846","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13846"}},"official":{"repos":["zhaoyi11/adaptive_bc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mocoda-model-based-counterfactual-data","slug":"mocoda-model-based-counterfactual-data","title":"MoCoDA: Model-based Counterfactual Data Augmentation","date":"2022-10-20","arxiv_id":"2210.11287","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mocoda-model-based-counterfactual-data#ran","syntology_url":"https://syntology.ai/paper/2210.11287","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.11287"}},"official":{"repos":["spitis/mocoda"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-policy-guided-imitation-approach-for","slug":"a-policy-guided-imitation-approach-for","title":"A Policy-Guided Imitation Approach for Offline Reinforcement Learning","date":"2022-10-15","arxiv_id":"2210.08323","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/a-policy-guided-imitation-approach-for#ran","syntology_url":"https://syntology.ai/paper/2210.08323","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.08323"}},"official":{"repos":["ryanxhr/por"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mutual-information-regularized-offline-1","slug":"mutual-information-regularized-offline-1","title":"Mutual Information Regularized Offline Reinforcement Learning","date":"2022-10-14","arxiv_id":"2210.07484","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mutual-information-regularized-offline-1#ran","syntology_url":"https://syntology.ai/paper/2210.07484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07484"}},"official":{"repos":["sail-sg/misa"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/semi-supervised-offline-reinforcement-1","slug":"semi-supervised-offline-reinforcement-1","title":"Semi-Supervised Offline Reinforcement Learning with Action-Free Trajectories","date":"2022-10-12","arxiv_id":"2210.06518","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/semi-supervised-offline-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2210.06518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06518"}},"official":{"repos":["facebookresearch/ssorl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/s2p-state-conditioned-image-synthesis-for","slug":"s2p-state-conditioned-image-synthesis-for","title":"S2P: State-conditioned Image Synthesis for Data Augmentation in Offline Reinforcement Learning","date":"2022-09-30","arxiv_id":"2209.15256","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/s2p-state-conditioned-image-synthesis-for#ran","syntology_url":"https://syntology.ai/paper/2209.15256","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.15256"}},"official":{"repos":["dsshim0125/s2p"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vip-towards-universal-visual-reward-and","slug":"vip-towards-universal-visual-reward-and","title":"VIP: Towards Universal Visual Reward and Representation via Value-Implicit Pre-Training","date":"2022-09-30","arxiv_id":"2210.00030","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vip-towards-universal-visual-reward-and#ran","syntology_url":"https://syntology.ai/paper/2210.00030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.00030"}},"official":{"repos":["facebookresearch/vip"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-planning-in-a-compact-latent-action","slug":"efficient-planning-in-a-compact-latent-action","title":"Efficient Planning in a Compact Latent Action Space","date":"2022-08-22","arxiv_id":"2208.10291","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-planning-in-a-compact-latent-action#ran","syntology_url":"https://syntology.ai/paper/2208.10291","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.10291"}},"official":{"repos":["ZhengyaoJiang/latentplan"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-policies-as-an-expressive-policy","slug":"diffusion-policies-as-an-expressive-policy","title":"Diffusion Policies as an Expressive Policy Class for Offline Reinforcement Learning","date":"2022-08-12","arxiv_id":"2208.06193","repositories_listed":3,"syntology":{"n":18,"n_ran":11,"n_constructed":5,"n_ran_checked":11,"n_instrument":0,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":10,"phrase":"11 ran (of which 5 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/diffusion-policies-as-an-expressive-policy#ran","syntology_url":"https://syntology.ai/paper/2208.06193","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.06193"}},"official":{"repos":["zhendong-wang/diffusion-policies-for-offline-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/discriminator-weighted-offline-imitation-1","slug":"discriminator-weighted-offline-imitation-1","title":"Discriminator-Weighted Offline Imitation Learning from Suboptimal Demonstrations","date":"2022-07-20","arxiv_id":"2207.10050","repositories_listed":2,"syntology":{"n":13,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":13,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/discriminator-weighted-offline-imitation-1#ran","syntology_url":"https://syntology.ai/paper/2207.10050","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.10050"}},"official":{"repos":["ryanxhr/dwbc"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/behavior-transformers-cloning-k-modes-with","slug":"behavior-transformers-cloning-k-modes-with","title":"Behavior Transformers: Cloning $k$ modes with one stone","date":"2022-06-22","arxiv_id":"2206.11251","repositories_listed":2,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/behavior-transformers-cloning-k-modes-with#ran","syntology_url":"https://syntology.ai/paper/2206.11251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.11251"}},"official":{"repos":["notmahi/bet"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/bootstrapped-transformer-for-offline","slug":"bootstrapped-transformer-for-offline","title":"Bootstrapped Transformer for Offline Reinforcement Learning","date":"2022-06-17","arxiv_id":"2206.08569","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bootstrapped-transformer-for-offline#ran","syntology_url":"https://syntology.ai/paper/2206.08569","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08569"}},"official":null}},{"url":"/paper/double-check-your-state-before-trusting-it","slug":"double-check-your-state-before-trusting-it","title":"Double Check Your State Before Trusting It: Confidence-Aware Bidirectional Offline Model-Based Imagination","date":"2022-06-16","arxiv_id":"2206.07989","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":1,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/double-check-your-state-before-trusting-it#ran","syntology_url":"https://syntology.ai/paper/2206.07989","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07989"}},"official":{"repos":["dmksjfl/CABI"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/regularizing-a-model-based-policy-stationary","slug":"regularizing-a-model-based-policy-stationary","title":"Regularizing a Model-based Policy Stationary Distribution to Stabilize Offline Reinforcement Learning","date":"2022-06-14","arxiv_id":"2206.07166","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regularizing-a-model-based-policy-stationary#ran","syntology_url":"https://syntology.ai/paper/2206.07166","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07166"}},"official":{"repos":["shentao-yang/sdm-gan_icml2022"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rorl-robust-offline-reinforcement-learning","slug":"rorl-robust-offline-reinforcement-learning","title":"RORL: Robust Offline Reinforcement Learning via Conservative Smoothing","date":"2022-06-06","arxiv_id":"2206.02829","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rorl-robust-offline-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2206.02829","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.02829"}},"official":{"repos":["yangrui2015/rorl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-rl-for-natural-language-generation","slug":"offline-rl-for-natural-language-generation","title":"Offline RL for Natural Language Generation with Implicit Language Q Learning","date":"2022-06-05","arxiv_id":"2206.11871","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/offline-rl-for-natural-language-generation#ran","syntology_url":"https://syntology.ai/paper/2206.11871","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.11871"}},"official":null}},{"url":"/paper/rambo-rl-robust-adversarial-model-based","slug":"rambo-rl-robust-adversarial-model-based","title":"RAMBO-RL: Robust Adversarial Model-Based Offline Reinforcement Learning","date":"2022-04-26","arxiv_id":"2204.12581","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rambo-rl-robust-adversarial-model-based#ran","syntology_url":"https://syntology.ai/paper/2204.12581","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.12581"}},"official":{"repos":["marc-rigter/rambo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/coptidice-offline-constrained-reinforcement-1","slug":"coptidice-offline-constrained-reinforcement-1","title":"COptiDICE: Offline Constrained Reinforcement Learning via Stationary Distribution Correction Estimation","date":"2022-04-19","arxiv_id":"2204.08957","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/coptidice-offline-constrained-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2204.08957","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.08957"}},"official":{"repos":["deepmind/constrained_optidice"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/chai-a-chatbot-ai-for-task-oriented-dialogue","slug":"chai-a-chatbot-ai-for-task-oriented-dialogue","title":"CHAI: A CHatbot AI for Task-Oriented Dialogue with Offline Reinforcement Learning","date":"2022-04-18","arxiv_id":"2204.08426","repositories_listed":2,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/chai-a-chatbot-ai-for-task-oriented-dialogue#ran","syntology_url":"https://syntology.ai/paper/2204.08426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.08426"}},"official":{"repos":["siddharthverma314/chai-naacl-2022"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cirs-bursting-filter-bubbles-by","slug":"cirs-bursting-filter-bubbles-by","title":"CIRS: Bursting Filter Bubbles by Counterfactual Interactive Recommender System","date":"2022-04-04","arxiv_id":"2204.01266","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cirs-bursting-filter-bubbles-by#ran","syntology_url":"https://syntology.ai/paper/2204.01266","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.01266"}},"official":{"repos":["chongminggao/cirs-codes"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/latent-variable-advantage-weighted-policy","slug":"latent-variable-advantage-weighted-policy","title":"Latent-Variable Advantage-Weighted Policy Optimization for Offline RL","date":"2022-03-16","arxiv_id":"2203.08949","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/latent-variable-advantage-weighted-policy#ran","syntology_url":"https://syntology.ai/paper/2203.08949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08949"}},"official":null}},{"url":"/paper/a-survey-on-offline-reinforcement-learning","slug":"a-survey-on-offline-reinforcement-learning","title":"A Survey on Offline Reinforcement Learning: Taxonomy, Review, and Open Problems","date":"2022-03-02","arxiv_id":"2203.01387","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-survey-on-offline-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2203.01387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.01387"}},"official":{"repos":["larocs/offline-rl-suvey"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vrl3-a-data-driven-framework-for-visual-deep","slug":"vrl3-a-data-driven-framework-for-visual-deep","title":"VRL3: A Data-Driven Framework for Visual Deep Reinforcement Learning","date":"2022-02-17","arxiv_id":"2202.10324","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":4,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/vrl3-a-data-driven-framework-for-visual-deep#ran","syntology_url":"https://syntology.ai/paper/2202.10324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.10324"}},"official":{"repos":["facebookresearch/drqv2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/flowformer-linearizing-transformers-with","slug":"flowformer-linearizing-transformers-with","title":"Flowformer: Linearizing Transformers with Conservation Flows","date":"2022-02-13","arxiv_id":"2202.06258","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/flowformer-linearizing-transformers-with#ran","syntology_url":"https://syntology.ai/paper/2202.06258","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.06258"}},"official":{"repos":["thuml/Flowformer"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-goal-conditioned-supervised-1","slug":"rethinking-goal-conditioned-supervised-1","title":"Rethinking Goal-conditioned Supervised Learning and Its Connection to Offline RL","date":"2022-02-09","arxiv_id":"2202.04478","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rethinking-goal-conditioned-supervised-1#ran","syntology_url":"https://syntology.ai/paper/2202.04478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.04478"}},"official":{"repos":["yangrui2015/awgcsl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/adversarially-trained-actor-critic-for","slug":"adversarially-trained-actor-critic-for","title":"Adversarially Trained Actor Critic for Offline Reinforcement Learning","date":"2022-02-05","arxiv_id":"2202.02446","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adversarially-trained-actor-critic-for#ran","syntology_url":"https://syntology.ai/paper/2202.02446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.02446"}},"official":{"repos":["microsoft/atac"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/don-t-change-the-algorithm-change-the-data","slug":"don-t-change-the-algorithm-change-the-data","title":"Don't Change the Algorithm, Change the Data: Exploratory Data for Offline Reinforcement Learning","date":"2022-01-31","arxiv_id":"2201.13425","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/don-t-change-the-algorithm-change-the-data#ran","syntology_url":"https://syntology.ai/paper/2201.13425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.13425"}},"official":{"repos":["denisyarats/exorl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-wikipedia-help-offline-reinforcement","slug":"can-wikipedia-help-offline-reinforcement","title":"Can Wikipedia Help Offline Reinforcement Learning?","date":"2022-01-28","arxiv_id":"2201.12122","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-wikipedia-help-offline-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2201.12122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12122"}},"official":{"repos":["machelreid/can-wikipedia-help-offline-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rvs-what-is-essential-for-offline-rl-via","slug":"rvs-what-is-essential-for-offline-rl-via","title":"RvS: What is Essential for Offline RL via Supervised Learning?","date":"2021-12-20","arxiv_id":"2112.10751","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rvs-what-is-essential-for-offline-rl-via#ran","syntology_url":"https://syntology.ai/paper/2112.10751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.10751"}},"official":{"repos":["scottemmons/rvs"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-on-policy-data-collection-for-data","slug":"robust-on-policy-data-collection-for-data","title":"Robust On-Policy Sampling for Data-Efficient Policy Evaluation in Reinforcement Learning","date":"2021-11-29","arxiv_id":"2111.14552","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-on-policy-data-collection-for-data#ran","syntology_url":"https://syntology.ai/paper/2111.14552","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.14552"}},"official":{"repos":["uoe-agents/robust_onpolicy_data_collection"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rlds-an-ecosystem-to-generate-share-and-use","slug":"rlds-an-ecosystem-to-generate-share-and-use","title":"RLDS: an Ecosystem to Generate, Share and Use Datasets in Reinforcement Learning","date":"2021-11-04","arxiv_id":"2111.02767","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rlds-an-ecosystem-to-generate-share-and-use#ran","syntology_url":"https://syntology.ai/paper/2111.02767","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.02767"}},"official":{"repos":["google-research/rlds"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/curriculum-offline-imitation-learning","slug":"curriculum-offline-imitation-learning","title":"Curriculum Offline Imitation Learning","date":"2021-11-03","arxiv_id":"2111.02056","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/curriculum-offline-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/2111.02056","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.02056"}},"official":{"repos":["apexrl/coil"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-driving-via-expert-guided-policy","slug":"safe-driving-via-expert-guided-policy","title":"Safe Driving via Expert Guided Policy Optimization","date":"2021-10-13","arxiv_id":"2110.06831","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safe-driving-via-expert-guided-policy#ran","syntology_url":"https://syntology.ai/paper/2110.06831","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06831"}},"official":{"repos":["decisionforce/EGPO"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-reinforcement-learning-with-implicit","slug":"offline-reinforcement-learning-with-implicit","title":"Offline Reinforcement Learning with Implicit Q-Learning","date":"2021-10-12","arxiv_id":"2110.06169","repositories_listed":17,"syntology":{"n":58,"n_ran":45,"n_constructed":23,"n_ran_checked":37,"n_instrument":8,"n_unverified":13,"n_honours":0,"n_violates":0,"n_no_contract":37,"n_pointer_only":23,"phrase":"45 ran (of which 23 constructed an object rather than computing a result; 37 with no instrument failure: 0 honoured, 0 violated, 37 with no contract checked; 8 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/offline-reinforcement-learning-with-implicit#ran","syntology_url":"https://syntology.ai/paper/2110.06169","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06169"}},"official":{"repos":["rail-berkeley/rlkit","ikostrikov/implicit_q_learning"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/beyond-pick-and-place-tackling-robotic","slug":"beyond-pick-and-place-tackling-robotic","title":"Beyond Pick-and-Place: Tackling Robotic Stacking of Diverse Shapes","date":"2021-10-12","arxiv_id":"2110.06192","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-pick-and-place-tackling-robotic#ran","syntology_url":"https://syntology.ai/paper/2110.06192","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06192"}},"official":{"repos":["deepmind/rgb_stacking"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/starformer-transformer-with-state-action-1","slug":"starformer-transformer-with-state-action-1","title":"StARformer: Transformer with State-Action-Reward Representations for Visual Reinforcement Learning","date":"2021-10-12","arxiv_id":"2110.06206","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/starformer-transformer-with-state-action-1#ran","syntology_url":"https://syntology.ai/paper/2110.06206","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06206"}},"official":{"repos":["elicassion/StARformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/uncertainty-based-offline-reinforcement","slug":"uncertainty-based-offline-reinforcement","title":"Uncertainty-Based Offline Reinforcement Learning with Diversified Q-Ensemble","date":"2021-10-04","arxiv_id":"2110.01548","repositories_listed":5,"syntology":{"n":21,"n_ran":13,"n_constructed":11,"n_ran_checked":12,"n_instrument":1,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":6,"phrase":"13 ran (of which 11 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/uncertainty-based-offline-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2110.01548","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.01548"}},"official":null}},{"url":"/paper/a-workflow-for-offline-model-free-robotic","slug":"a-workflow-for-offline-model-free-robotic","title":"A Workflow for Offline Model-Free Robotic Reinforcement Learning","date":"2021-09-22","arxiv_id":"2109.10813","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-workflow-for-offline-model-free-robotic#ran","syntology_url":"https://syntology.ai/paper/2109.10813","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.10813"}},"official":null}},{"url":"/paper/conservative-offline-distributional","slug":"conservative-offline-distributional","title":"Conservative Offline Distributional Reinforcement Learning","date":"2021-07-12","arxiv_id":"2107.06106","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/conservative-offline-distributional#ran","syntology_url":"https://syntology.ai/paper/2107.06106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.06106"}},"official":{"repos":["JasonMa2016/CODAC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-minimalist-approach-to-offline","slug":"a-minimalist-approach-to-offline","title":"A Minimalist Approach to Offline Reinforcement Learning","date":"2021-06-12","arxiv_id":"2106.06860","repositories_listed":8,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-minimalist-approach-to-offline#ran","syntology_url":"https://syntology.ai/paper/2106.06860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.06860"}},"official":{"repos":["sfujim/TD3_BC"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/reinforcement-learning-as-one-big-sequence","slug":"reinforcement-learning-as-one-big-sequence","title":"Offline Reinforcement Learning as One Big Sequence Modeling Problem","date":"2021-06-03","arxiv_id":"2106.02039","repositories_listed":2,"syntology":{"n":28,"n_ran":23,"n_constructed":0,"n_ran_checked":22,"n_instrument":1,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":21,"n_pointer_only":1,"phrase":"23 ran (of which 0 constructed an object rather than computing a result; 22 with no instrument failure: 1 honoured, 0 violated, 21 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/reinforcement-learning-as-one-big-sequence#ran","syntology_url":"https://syntology.ai/paper/2106.02039","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.02039"}},"official":{"repos":["JannerM/trajectory-transformer"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":5,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/decision-transformer-reinforcement-learning","slug":"decision-transformer-reinforcement-learning","title":"Decision Transformer: Reinforcement Learning via Sequence Modeling","date":"2021-06-02","arxiv_id":"2106.01345","repositories_listed":20,"syntology":{"n":26,"n_ran":17,"n_constructed":10,"n_ran_checked":13,"n_instrument":4,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":8,"phrase":"17 ran (of which 10 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 4 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/decision-transformer-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2106.01345","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.01345"}},"official":{"repos":["kzl/decision-transformer"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/human-centric-dialog-training-via-offline","slug":"human-centric-dialog-training-via-offline","title":"Human-centric Dialog Training via Offline Reinforcement Learning","date":"2020-10-12","arxiv_id":"2010.05848","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/human-centric-dialog-training-via-offline#ran","syntology_url":"https://syntology.ai/paper/2010.05848","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.05848"}},"official":{"repos":["natashamjaques/neural_chat"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/efficient-fully-offline-meta-reinforcement","slug":"efficient-fully-offline-meta-reinforcement","title":"FOCAL: Efficient Fully-Offline Meta-Reinforcement Learning via Distance Metric Learning and Behavior Regularization","date":"2020-10-02","arxiv_id":"2010.01112","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/efficient-fully-offline-meta-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2010.01112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.01112"}},"official":{"repos":["FOCAL-ICLR/FOCAL-ICLR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-attention-with-performers","slug":"rethinking-attention-with-performers","title":"Rethinking Attention with Performers","date":"2020-09-30","arxiv_id":"2009.14794","repositories_listed":7,"syntology":{"n":16,"n_ran":11,"n_constructed":3,"n_ran_checked":7,"n_instrument":4,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"11 ran (of which 3 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/rethinking-attention-with-performers#ran","syntology_url":"https://syntology.ai/paper/2009.14794","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.14794"}},"official":{"repos":["google-research/google-research"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/offline-meta-reinforcement-learning-with","slug":"offline-meta-reinforcement-learning-with","title":"Offline Meta-Reinforcement Learning with Advantage Weighting","date":"2020-08-13","arxiv_id":"2008.06043","repositories_listed":2,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/offline-meta-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2008.06043","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.06043"}},"official":{"repos":["eric-mitchell/macaw"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/transformers-are-rnns-fast-autoregressive","slug":"transformers-are-rnns-fast-autoregressive","title":"Transformers are RNNs: Fast Autoregressive Transformers with Linear Attention","date":"2020-06-29","arxiv_id":"2006.16236","repositories_listed":8,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":2,"n_no_contract":3,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/transformers-are-rnns-fast-autoregressive#ran","syntology_url":"https://syntology.ai/paper/2006.16236","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.16236"}},"official":{"repos":["idiap/fast-transformers"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/conservative-q-learning-for-offline","slug":"conservative-q-learning-for-offline","title":"Conservative Q-Learning for Offline Reinforcement Learning","date":"2020-06-08","arxiv_id":"2006.04779","repositories_listed":18,"syntology":{"n":34,"n_ran":31,"n_constructed":5,"n_ran_checked":28,"n_instrument":3,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":26,"n_pointer_only":19,"phrase":"31 ran (of which 5 constructed an object rather than computing a result; 28 with no instrument failure: 2 honoured, 0 violated, 26 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/conservative-q-learning-for-offline#ran","syntology_url":"https://syntology.ai/paper/2006.04779","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04779"}},"official":{"repos":["aviralkumar2907/CQL"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/deployment-efficient-reinforcement-learning","slug":"deployment-efficient-reinforcement-learning","title":"Deployment-Efficient Reinforcement Learning via Model-Based Offline Optimization","date":"2020-06-05","arxiv_id":"2006.03647","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deployment-efficient-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2006.03647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.03647"}},"official":{"repos":["matsuolab/BREMEN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mopo-model-based-offline-policy-optimization","slug":"mopo-model-based-offline-policy-optimization","title":"MOPO: Model-based Offline Policy Optimization","date":"2020-05-27","arxiv_id":"2005.13239","repositories_listed":6,"syntology":{"n":8,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mopo-model-based-offline-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2005.13239","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.13239"}},"official":{"repos":["tianheyu927/mopo"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/morel-model-based-offline-reinforcement","slug":"morel-model-based-offline-reinforcement","title":"MOReL : Model-Based Offline Reinforcement Learning","date":"2020-05-12","arxiv_id":"2005.05951","repositories_listed":2,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/morel-model-based-offline-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2005.05951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.05951"}},"official":null}},{"url":"/paper/datasets-for-data-driven-reinforcement","slug":"datasets-for-data-driven-reinforcement","title":"D4RL: Datasets for Deep Data-Driven Reinforcement Learning","date":"2020-04-15","arxiv_id":"2004.07219","repositories_listed":7,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/datasets-for-data-driven-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2004.07219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.07219"}},"official":{"repos":["rail-berkeley/d4rl","rail-berkeley/offline_rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/reformer-the-efficient-transformer-1","slug":"reformer-the-efficient-transformer-1","title":"Reformer: The Efficient Transformer","date":"2020-01-13","arxiv_id":"2001.04451","repositories_listed":10,"syntology":{"n":8,"n_ran":6,"n_constructed":1,"n_ran_checked":1,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/reformer-the-efficient-transformer-1#ran","syntology_url":"https://syntology.ai/paper/2001.04451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.04451"}},"official":{"repos":["google/trax"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}}],"record_sha256":"aaa89653fe08e6995a9894d778fc72cfa6845e4b179bec8bf025c99f4da81e4f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}