{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/offline-rl/papers/ran/1","list_of":"/task/offline-rl","task":"Offline RL","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":2,"rows_per_page":100,"rows":[1,100],"of":164,"counts":{"archive_papers_tagged":755,"with_a_code_link":310,"where_syntology_ran_a_sample":164,"not_listed_spam_title":0,"listed":755,"listed_where_code_ran":164,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":139,"every_run_a_failure_of_syntologys_instrument":25,"listed_with_a_run_with_no_instrument_failure":139,"listed_every_run_a_failure_of_syntologys_instrument":25,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/offline-rl/papers/ran/1","prev":null,"next":"/task/offline-rl/papers/ran/2","papers":[{"url":"/paper/flow-based-single-step-completion-for","slug":"flow-based-single-step-completion-for","title":"Flow-Based Single-Step Completion for Efficient and Expressive Policy Learning","date":"2025-06-26","arxiv_id":"2506.21427","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/flow-based-single-step-completion-for#ran","syntology_url":"https://syntology.ai/paper/2506.21427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.21427"}},"official":null}},{"url":"/paper/dr-sac-distributionally-robust-soft-actor","slug":"dr-sac-distributionally-robust-soft-actor","title":"DR-SAC: Distributionally Robust Soft Actor-Critic for Reinforcement Learning under Uncertainty","date":"2025-06-14","arxiv_id":"2506.12622","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/dr-sac-distributionally-robust-soft-actor#ran","syntology_url":"https://syntology.ai/paper/2506.12622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.12622"}},"official":{"repos":["lemutisme/dr-sac"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-rl-with-smooth-ood-generalization-in","slug":"offline-rl-with-smooth-ood-generalization-in","title":"Offline RL with Smooth OOD Generalization in Convex Hull and its Neighborhood","date":"2025-06-10","arxiv_id":"2506.08417","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/offline-rl-with-smooth-ood-generalization-in#ran","syntology_url":"https://syntology.ai/paper/2506.08417","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08417"}},"official":{"repos":["yqpqry/sqog"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/2506-08460","slug":"2506-08460","title":"MOBODY: Model Based Off-Dynamics Offline Reinforcement Learning","date":"2025-06-10","arxiv_id":"2506.08460","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":3,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/2506-08460#ran","syntology_url":"https://syntology.ai/paper/2506.08460","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08460"}},"official":{"repos":["guoyihonggyh/mobody-model-based-off-dynamics-offline-reinforcement-learning"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-trust-bellman-updates-selective","slug":"learning-to-trust-bellman-updates-selective","title":"Learning to Trust Bellman Updates: Selective State-Adaptive Regularization for Offline RL","date":"2025-05-26","arxiv_id":"2505.19923","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":2,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":2,"n_pointer_only":7,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 2 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-trust-bellman-updates-selective#ran","syntology_url":"https://syntology.ai/paper/2505.19923","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19923"}},"official":{"repos":["qinwenluo/ssar"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vipo-value-function-inconsistency-penalized","slug":"vipo-value-function-inconsistency-penalized","title":"VIPO: Value Function Inconsistency Penalized Offline Reinforcement Learning","date":"2025-04-16","arxiv_id":"2504.11944","repositories_listed":0,"syntology":{"n":14,"n_ran":10,"n_constructed":9,"n_ran_checked":9,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":14,"phrase":"10 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vipo-value-function-inconsistency-penalized#ran","syntology_url":"https://syntology.ai/paper/2504.11944","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11944"}},"official":null}},{"url":"/paper/active-advantage-aligned-online-reinforcement","slug":"active-advantage-aligned-online-reinforcement","title":"Active Advantage-Aligned Online Reinforcement Learning with Offline Data","date":"2025-02-11","arxiv_id":"2502.07937","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/active-advantage-aligned-online-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2502.07937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.07937"}},"official":{"repos":["xuefeng-cs/a3rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flow-q-learning","slug":"flow-q-learning","title":"Flow Q-Learning","date":"2025-02-04","arxiv_id":"2502.02538","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_constructed":4,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/flow-q-learning#ran","syntology_url":"https://syntology.ai/paper/2502.02538","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.02538"}},"official":{"repos":["seohongpark/fql"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/optimistic-critic-reconstruction-and","slug":"optimistic-critic-reconstruction-and","title":"Optimistic Critic Reconstruction and Constrained Fine-Tuning for General Offline-to-Online RL","date":"2024-12-25","arxiv_id":"2412.18855","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":3,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/optimistic-critic-reconstruction-and#ran","syntology_url":"https://syntology.ai/paper/2412.18855","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.18855"}},"official":{"repos":["QinwenLuo/OCR-CFT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/constraint-adaptive-policy-switching-for","slug":"constraint-adaptive-policy-switching-for","title":"Constraint-Adaptive Policy Switching for Offline Safe Reinforcement Learning","date":"2024-12-25","arxiv_id":"2412.18946","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/constraint-adaptive-policy-switching-for#ran","syntology_url":"https://syntology.ai/paper/2412.18946","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.18946"}},"official":{"repos":["yassinech/caps"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-reinforcement-learning-for-llm-multi","slug":"offline-reinforcement-learning-for-llm-multi","title":"Offline Reinforcement Learning for LLM Multi-Step Reasoning","date":"2024-12-20","arxiv_id":"2412.16145","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/offline-reinforcement-learning-for-llm-multi#ran","syntology_url":"https://syntology.ai/paper/2412.16145","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.16145"}},"official":{"repos":["jwhj/oreo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/are-expressive-models-truly-necessary-for","slug":"are-expressive-models-truly-necessary-for","title":"Are Expressive Models Truly Necessary for Offline RL?","date":"2024-12-15","arxiv_id":"2412.11253","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/are-expressive-models-truly-necessary-for#ran","syntology_url":"https://syntology.ai/paper/2412.11253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.11253"}},"official":{"repos":["imoneoi/RSP_JAX"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/latent-safety-constrained-policy-approach-for","slug":"latent-safety-constrained-policy-approach-for","title":"Latent Safety-Constrained Policy Approach for Safe Offline Reinforcement Learning","date":"2024-12-11","arxiv_id":"2412.08794","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/latent-safety-constrained-policy-approach-for#ran","syntology_url":"https://syntology.ai/paper/2412.08794","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.08794"}},"official":{"repos":["PrajwalKoirala/LSPC-Safe-Offline-RL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-online-reinforcement-learning-fine","slug":"efficient-online-reinforcement-learning-fine","title":"Efficient Online Reinforcement Learning Fine-Tuning Need Not Retain Offline Data","date":"2024-12-10","arxiv_id":"2412.07762","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-online-reinforcement-learning-fine#ran","syntology_url":"https://syntology.ai/paper/2412.07762","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.07762"}},"official":{"repos":["zhouzypaul/wsrl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-offline-reinforcement-learning-with-1","slug":"robust-offline-reinforcement-learning-with-1","title":"Robust Offline Reinforcement Learning with Linearly Structured $f$-Divergence Regularization","date":"2024-11-27","arxiv_id":"2411.18612","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/robust-offline-reinforcement-learning-with-1#ran","syntology_url":"https://syntology.ai/paper/2411.18612","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.18612"}},"official":null}},{"url":"/paper/continual-task-learning-through-adaptive","slug":"continual-task-learning-through-adaptive","title":"Continual Task Learning through Adaptive Policy Self-Composition","date":"2024-11-18","arxiv_id":"2411.11364","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":4,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/continual-task-learning-through-adaptive#ran","syntology_url":"https://syntology.ai/paper/2411.11364","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.11364"}},"official":{"repos":["charleshsc/CompoFormer"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/doubly-mild-generalization-for-offline","slug":"doubly-mild-generalization-for-offline","title":"Doubly Mild Generalization for Offline Reinforcement Learning","date":"2024-11-12","arxiv_id":"2411.07934","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/doubly-mild-generalization-for-offline#ran","syntology_url":"https://syntology.ai/paper/2411.07934","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.07934"}},"official":{"repos":["maoyixiu/dmg"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/uncertainty-based-offline-variational","slug":"uncertainty-based-offline-variational","title":"Uncertainty-based Offline Variational Bayesian Reinforcement Learning for Robustness under Diverse Data Corruptions","date":"2024-11-01","arxiv_id":"2411.00465","repositories_listed":1,"syntology":{"n":13,"n_ran":4,"n_constructed":2,"n_ran_checked":2,"n_instrument":2,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/uncertainty-based-offline-variational#ran","syntology_url":"https://syntology.ai/paper/2411.00465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00465"}},"official":{"repos":["MIRALab-USTC/RL-TRACER"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/networkgym-reinforcement-learning","slug":"networkgym-reinforcement-learning","title":"NetworkGym: Reinforcement Learning Environments for Multi-Access Traffic Management in Network Simulation","date":"2024-10-30","arxiv_id":"2411.04138","repositories_listed":1,"syntology":{"n":18,"n_ran":15,"n_constructed":2,"n_ran_checked":12,"n_instrument":3,"n_unverified":3,"n_honours":4,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"15 ran (of which 2 constructed an object rather than computing a result; 12 with no instrument failure: 4 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/networkgym-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2411.04138","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.04138"}},"official":{"repos":["hmomin/networkgym"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":2,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/longreward-improving-long-context-large","slug":"longreward-improving-long-context-large","title":"LongReward: Improving Long-context Large Language Models with AI Feedback","date":"2024-10-28","arxiv_id":"2410.21252","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longreward-improving-long-context-large#ran","syntology_url":"https://syntology.ai/paper/2410.21252","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21252"}},"official":{"repos":["THUDM/LongReward"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/offline-reinforcement-learning-with-ood-state","slug":"offline-reinforcement-learning-with-ood-state","title":"Offline Reinforcement Learning with OOD State Correction and OOD Action Suppression","date":"2024-10-25","arxiv_id":"2410.19400","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/offline-reinforcement-learning-with-ood-state#ran","syntology_url":"https://syntology.ai/paper/2410.19400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.19400"}},"official":{"repos":["MAOYIXIU/SCAS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-versatile-skills-with-curriculum","slug":"learning-versatile-skills-with-curriculum","title":"Learning Versatile Skills with Curriculum Masking","date":"2024-10-23","arxiv_id":"2410.17744","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-versatile-skills-with-curriculum#ran","syntology_url":"https://syntology.ai/paper/2410.17744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17744"}},"official":{"repos":["yaotang23/currmask"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bayes-adaptive-monte-carlo-tree-search-for","slug":"bayes-adaptive-monte-carlo-tree-search-for","title":"Bayes Adaptive Monte Carlo Tree Search for Offline Model-based Reinforcement Learning","date":"2024-10-15","arxiv_id":"2410.11234","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bayes-adaptive-monte-carlo-tree-search-for#ran","syntology_url":"https://syntology.ai/paper/2410.11234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11234"}},"official":{"repos":["lucascjysdl/offline-rl-kit"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-offline-model-based-rl-via-jointly","slug":"scaling-offline-model-based-rl-via-jointly","title":"Scaling Offline Model-Based RL via Jointly-Optimized World-Action Model Pretraining","date":"2024-10-01","arxiv_id":"2410.00564","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/scaling-offline-model-based-rl-via-jointly#ran","syntology_url":"https://syntology.ai/paper/2410.00564","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00564"}},"official":{"repos":["cjreinforce/jowa"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/dmc-vb-a-benchmark-for-representation","slug":"dmc-vb-a-benchmark-for-representation","title":"DMC-VB: A Benchmark for Representation Learning for Control with Visual Distractors","date":"2024-09-26","arxiv_id":"2409.18330","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dmc-vb-a-benchmark-for-representation#ran","syntology_url":"https://syntology.ai/paper/2409.18330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.18330"}},"official":{"repos":["google-deepmind/dmc_vision_benchmark"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hokoff-real-game-dataset-from-honor-of-kings-1","slug":"hokoff-real-game-dataset-from-honor-of-kings-1","title":"Hokoff: Real Game Dataset from Honor of Kings and its Offline Reinforcement Learning Benchmarks","date":"2024-08-20","arxiv_id":"2408.10556","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hokoff-real-game-dataset-from-honor-of-kings-1#ran","syntology_url":"https://syntology.ai/paper/2408.10556","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.10556"}},"official":{"repos":["tencent-ailab/hokoff"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/digirl-training-in-the-wild-device-control","slug":"digirl-training-in-the-wild-device-control","title":"DigiRL: Training In-The-Wild Device-Control Agents with Autonomous Reinforcement Learning","date":"2024-06-14","arxiv_id":"2406.11896","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/digirl-training-in-the-wild-device-control#ran","syntology_url":"https://syntology.ai/paper/2406.11896","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11896"}},"official":null}},{"url":"/paper/is-value-learning-really-the-main-bottleneck","slug":"is-value-learning-really-the-main-bottleneck","title":"Is Value Learning Really the Main Bottleneck in Offline RL?","date":"2024-06-13","arxiv_id":"2406.09329","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":6,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/is-value-learning-really-the-main-bottleneck#ran","syntology_url":"https://syntology.ai/paper/2406.09329","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09329"}},"official":null}},{"url":"/paper/decision-mamba-a-multi-grained-state-space","slug":"decision-mamba-a-multi-grained-state-space","title":"Decision Mamba: A Multi-Grained State Space Model with Self-Evolution Regularization for Offline RL","date":"2024-06-08","arxiv_id":"2406.05427","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":4,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/decision-mamba-a-multi-grained-state-space#ran","syntology_url":"https://syntology.ai/paper/2406.05427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05427"}},"official":{"repos":["aopolin-lv/DecisionMamba"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"url":"/paper/diffusion-policies-creating-a-trust-region","slug":"diffusion-policies-creating-a-trust-region","title":"Diffusion Policies creating a Trust Region for Offline Reinforcement Learning","date":"2024-05-30","arxiv_id":"2405.19690","repositories_listed":1,"syntology":{"n":20,"n_ran":17,"n_constructed":0,"n_ran_checked":11,"n_instrument":6,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":20,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/diffusion-policies-creating-a-trust-region#ran","syntology_url":"https://syntology.ai/paper/2405.19690","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19690"}},"official":{"repos":["tianyucodings/diffusion_trusted_q_learning"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-in-dynamic-treatment","slug":"reinforcement-learning-in-dynamic-treatment","title":"Reinforcement Learning in Dynamic Treatment Regimes Needs Critical Reexamination","date":"2024-05-28","arxiv_id":"2405.18556","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reinforcement-learning-in-dynamic-treatment#ran","syntology_url":"https://syntology.ai/paper/2405.18556","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.18556"}},"official":{"repos":["gilesluo/reassessdtr"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/gta-generative-trajectory-augmentation-with","slug":"gta-generative-trajectory-augmentation-with","title":"GTA: Generative Trajectory Augmentation with Guidance for Offline Reinforcement Learning","date":"2024-05-27","arxiv_id":"2405.16907","repositories_listed":1,"syntology":{"n":37,"n_ran":34,"n_constructed":0,"n_ran_checked":29,"n_instrument":5,"n_unverified":3,"n_honours":3,"n_violates":4,"n_no_contract":22,"n_pointer_only":5,"phrase":"34 ran (of which 0 constructed an object rather than computing a result; 29 with no instrument failure: 3 honoured, 4 violated, 22 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/gta-generative-trajectory-augmentation-with#ran","syntology_url":"https://syntology.ai/paper/2405.16907","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16907"}},"official":{"repos":["jaewoopudding/gta"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":3,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/any-step-dynamics-model-improves-future","slug":"any-step-dynamics-model-improves-future","title":"Any-step Dynamics Model Improves Future Predictions for Online and Offline Reinforcement Learning","date":"2024-05-27","arxiv_id":"2405.17031","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/any-step-dynamics-model-improves-future#ran","syntology_url":"https://syntology.ai/paper/2405.17031","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17031"}},"official":{"repos":["HxLyn3/ADMPO"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/q-value-regularized-transformer-for-offline","slug":"q-value-regularized-transformer-for-offline","title":"Q-value Regularized Transformer for Offline Reinforcement Learning","date":"2024-05-27","arxiv_id":"2405.17098","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/q-value-regularized-transformer-for-offline#ran","syntology_url":"https://syntology.ai/paper/2405.17098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17098"}},"official":null}},{"url":"/paper/generating-code-world-models-with-large","slug":"generating-code-world-models-with-large","title":"Generating Code World Models with Large Language Models Guided by Monte Carlo Tree Search","date":"2024-05-24","arxiv_id":"2405.15383","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generating-code-world-models-with-large#ran","syntology_url":"https://syntology.ai/paper/2405.15383","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15383"}},"official":{"repos":["nicoladainese96/code-world-models"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-reinforcement-learning-from-datasets","slug":"offline-reinforcement-learning-from-datasets","title":"Offline Reinforcement Learning from Datasets with Structured Non-Stationarity","date":"2024-05-23","arxiv_id":"2405.14114","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/offline-reinforcement-learning-from-datasets#ran","syntology_url":"https://syntology.ai/paper/2405.14114","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14114"}},"official":{"repos":["johannesack/offlinerlstructurednonstationarity"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/is-mamba-compatible-with-trajectory","slug":"is-mamba-compatible-with-trajectory","title":"Is Mamba Compatible with Trajectory Optimization in Offline Reinforcement Learning?","date":"2024-05-20","arxiv_id":"2405.12094","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/is-mamba-compatible-with-trajectory#ran","syntology_url":"https://syntology.ai/paper/2405.12094","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.12094"}},"official":{"repos":["AndssY/DeMa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/reinformer-max-return-sequence-modeling-for","slug":"reinformer-max-return-sequence-modeling-for","title":"Reinformer: Max-Return Sequence Modeling for Offline RL","date":"2024-05-14","arxiv_id":"2405.08740","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/reinformer-max-return-sequence-modeling-for#ran","syntology_url":"https://syntology.ai/paper/2405.08740","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.08740"}},"official":{"repos":["dragon-zhuang/reinformer"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ltldog-satisfying-temporally-extended","slug":"ltldog-satisfying-temporally-extended","title":"LTLDoG: Satisfying Temporally-Extended Symbolic Constraints for Safe Diffusion-based Planning","date":"2024-05-07","arxiv_id":"2405.04235","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ltldog-satisfying-temporally-extended#ran","syntology_url":"https://syntology.ai/paper/2405.04235","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.04235"}},"official":{"repos":["clear-nus/ltldog"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/compositional-conservatism-a-transductive","slug":"compositional-conservatism-a-transductive","title":"Compositional Conservatism: A Transductive Approach in Offline Reinforcement Learning","date":"2024-04-06","arxiv_id":"2404.04682","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":2,"n_ran_checked":2,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/compositional-conservatism-a-transductive#ran","syntology_url":"https://syntology.ai/paper/2404.04682","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04682"}},"official":{"repos":["runamu/compositional-conservatism"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-zero-shot-reinforcement-learning","slug":"unsupervised-zero-shot-reinforcement-learning","title":"Unsupervised Zero-Shot Reinforcement Learning via Functional Reward Encodings","date":"2024-02-27","arxiv_id":"2402.17135","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":1,"n_no_contract":6,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/unsupervised-zero-shot-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2402.17135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17135"}},"official":{"repos":["kvfrans/fre"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/stitching-sub-trajectories-with-conditional","slug":"stitching-sub-trajectories-with-conditional","title":"Stitching Sub-Trajectories with Conditional Diffusion Model for Goal-Conditioned Offline RL","date":"2024-02-11","arxiv_id":"2402.07226","repositories_listed":1,"syntology":{"n":29,"n_ran":20,"n_constructed":0,"n_ran_checked":14,"n_instrument":6,"n_unverified":9,"n_honours":2,"n_violates":0,"n_no_contract":12,"n_pointer_only":29,"phrase":"20 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 2 honoured, 0 violated, 12 with no contract checked; 6 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/stitching-sub-trajectories-with-conditional#ran","syntology_url":"https://syntology.ai/paper/2402.07226","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07226"}},"official":{"repos":["rlatjddbs/ssd"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/seabo-a-simple-search-based-method-for","slug":"seabo-a-simple-search-based-method-for","title":"SEABO: A Simple Search-Based Method for Offline Imitation Learning","date":"2024-02-06","arxiv_id":"2402.03807","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/seabo-a-simple-search-based-method-for#ran","syntology_url":"https://syntology.ai/paper/2402.03807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03807"}},"official":{"repos":["dmksjfl/seabo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/entropy-regularized-diffusion-policy-with-q","slug":"entropy-regularized-diffusion-policy-with-q","title":"Entropy-regularized Diffusion Policy with Q-Ensembles for Offline Reinforcement Learning","date":"2024-02-06","arxiv_id":"2402.04080","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/entropy-regularized-diffusion-policy-with-q#ran","syntology_url":"https://syntology.ai/paper/2402.04080","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.04080"}},"official":{"repos":["ruoqizzz/entropy-offlineRL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/value-aided-conditional-supervised-learning","slug":"value-aided-conditional-supervised-learning","title":"Adaptive $Q$-Aid for Conditional Supervised Learning in Offline Reinforcement Learning","date":"2024-02-03","arxiv_id":"2402.02017","repositories_listed":0,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/value-aided-conditional-supervised-learning#ran","syntology_url":"https://syntology.ai/paper/2402.02017","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02017"}},"official":null}},{"url":"/paper/odice-revealing-the-mystery-of-distribution","slug":"odice-revealing-the-mystery-of-distribution","title":"ODICE: Revealing the Mystery of Distribution Correction Estimation via Orthogonal-gradient Update","date":"2024-02-01","arxiv_id":"2402.00348","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":1,"n_ran_checked":5,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/odice-revealing-the-mystery-of-distribution#ran","syntology_url":"https://syntology.ai/paper/2402.00348","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.00348"}},"official":{"repos":["maoliyuan/odice-pytorch"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":1,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-offline-reinforcement-learning-with","slug":"safe-offline-reinforcement-learning-with","title":"Safe Offline Reinforcement Learning with Feasibility-Guided Diffusion Model","date":"2024-01-19","arxiv_id":"2401.10700","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":2,"n_no_contract":5,"n_pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 2 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safe-offline-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2401.10700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10700"}},"official":{"repos":["zhengyinan-air/fisor"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/diffclone-enhanced-behaviour-cloning-in","slug":"diffclone-enhanced-behaviour-cloning-in","title":"DiffClone: Enhanced Behaviour Cloning in Robotics with Diffusion-Driven Policy Learning","date":"2024-01-17","arxiv_id":"2401.09243","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diffclone-enhanced-behaviour-cloning-in#ran","syntology_url":"https://syntology.ai/paper/2401.09243","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09243"}},"official":{"repos":["sirabas369/diffclone"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-from-sparse-offline-datasets-via","slug":"learning-from-sparse-offline-datasets-via","title":"Learning from Sparse Offline Datasets via Conservative Density Estimation","date":"2024-01-16","arxiv_id":"2401.08819","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-from-sparse-offline-datasets-via#ran","syntology_url":"https://syntology.ai/paper/2401.08819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.08819"}},"official":{"repos":["czp16/cde-offline-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/spqr-controlling-q-ensemble-independence-with-1","slug":"spqr-controlling-q-ensemble-independence-with-1","title":"SPQR: Controlling Q-ensemble Independence with Spiked Random Model for Reinforcement Learning","date":"2024-01-06","arxiv_id":"2401.03137","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/spqr-controlling-q-ensemble-independence-with-1#ran","syntology_url":"https://syntology.ai/paper/2401.03137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.03137"}},"official":{"repos":["dohyeoklee/SPQR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-regularized-offline-multi-objective","slug":"policy-regularized-offline-multi-objective","title":"Policy-regularized Offline Multi-objective Reinforcement Learning","date":"2024-01-04","arxiv_id":"2401.02244","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/policy-regularized-offline-multi-objective#ran","syntology_url":"https://syntology.ai/paper/2401.02244","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.02244"}},"official":{"repos":["qianlin04/prmorl"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/critic-guided-decision-transformer-for","slug":"critic-guided-decision-transformer-for","title":"Critic-Guided Decision Transformer for Offline Reinforcement Learning","date":"2023-12-21","arxiv_id":"2312.13716","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/critic-guided-decision-transformer-for#ran","syntology_url":"https://syntology.ai/paper/2312.13716","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.13716"}},"official":{"repos":["sharkwyf/cgdt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/the-generalization-gap-in-offline","slug":"the-generalization-gap-in-offline","title":"The Generalization Gap in Offline Reinforcement Learning","date":"2023-12-10","arxiv_id":"2312.05742","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/the-generalization-gap-in-offline#ran","syntology_url":"https://syntology.ai/paper/2312.05742","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.05742"}},"official":{"repos":["facebookresearch/gen_dgrl"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-data-enhanced-on-policy-policy","slug":"offline-data-enhanced-on-policy-policy","title":"Offline Data Enhanced On-Policy Policy Gradient with Provable Guarantees","date":"2023-11-14","arxiv_id":"2311.08384","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/offline-data-enhanced-on-policy-policy#ran","syntology_url":"https://syntology.ai/paper/2311.08384","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08384"}},"official":{"repos":["yifeizhou02/hnpg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/unleashing-the-power-of-pre-trained-language","slug":"unleashing-the-power-of-pre-trained-language","title":"Unleashing the Power of Pre-trained Language Models for Offline Reinforcement Learning","date":"2023-10-31","arxiv_id":"2310.20587","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unleashing-the-power-of-pre-trained-language#ran","syntology_url":"https://syntology.ai/paper/2310.20587","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20587"}},"official":{"repos":["srzer/LaMo-2023"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-offline-policy-evaluation-and","slug":"robust-offline-policy-evaluation-and","title":"Robust Offline Reinforcement learning with Heavy-Tailed Rewards","date":"2023-10-28","arxiv_id":"2310.18715","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-offline-policy-evaluation-and#ran","syntology_url":"https://syntology.ai/paper/2310.18715","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.18715"}},"official":{"repos":["mamba413/room"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/corruption-robust-offline-reinforcement-2","slug":"corruption-robust-offline-reinforcement-2","title":"Corruption-Robust Offline Reinforcement Learning with General Function Approximation","date":"2023-10-23","arxiv_id":"2310.14550","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/corruption-robust-offline-reinforcement-2#ran","syntology_url":"https://syntology.ai/paper/2310.14550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14550"}},"official":{"repos":["yangrui2015/uwmsg"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-robust-offline-reinforcement-learning","slug":"towards-robust-offline-reinforcement-learning","title":"Towards Robust Offline Reinforcement Learning under Diverse Data Corruption","date":"2023-10-19","arxiv_id":"2310.12955","repositories_listed":2,"syntology":{"n":9,"n_ran":5,"n_constructed":2,"n_ran_checked":3,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":9,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/towards-robust-offline-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2310.12955","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12955"}},"official":{"repos":["yangrui2015/riql"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/offline-retraining-for-online-rl-decoupled","slug":"offline-retraining-for-online-rl-decoupled","title":"Offline Retraining for Online RL: Decoupled Policy Learning to Mitigate Exploration Bias","date":"2023-10-12","arxiv_id":"2310.08558","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/offline-retraining-for-online-rl-decoupled#ran","syntology_url":"https://syntology.ai/paper/2310.08558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08558"}},"official":{"repos":["MaxSobolMark/OOO"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/diffcps-diffusion-model-based-constrained","slug":"diffcps-diffusion-model-based-constrained","title":"DiffCPS: Diffusion Model based Constrained Policy Search for Offline Reinforcement Learning","date":"2023-10-09","arxiv_id":"2310.05333","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":4,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"9 ran (of which 4 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/diffcps-diffusion-model-based-constrained#ran","syntology_url":"https://syntology.ai/paper/2310.05333","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05333"}},"official":{"repos":["felix-thu/DiffCPS"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":4,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-predicting-and-better-resolving-1","slug":"understanding-predicting-and-better-resolving-1","title":"Understanding, Predicting and Better Resolving Q-Value Divergence in Offline-RL","date":"2023-10-06","arxiv_id":"2310.04411","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":4,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 4 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/understanding-predicting-and-better-resolving-1#ran","syntology_url":"https://syntology.ai/paper/2310.04411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04411"}},"official":{"repos":["yueyang130/seem"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-uniform-sampling-offline-reinforcement-1","slug":"beyond-uniform-sampling-offline-reinforcement-1","title":"Beyond Uniform Sampling: Offline Reinforcement Learning with Imbalanced Datasets","date":"2023-10-06","arxiv_id":"2310.04413","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/beyond-uniform-sampling-offline-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2310.04413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04413"}},"official":{"repos":["Improbable-AI/dw-offline-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-robust-offline-to-online","slug":"towards-robust-offline-to-online","title":"Towards Robust Offline-to-Online Reinforcement Learning via Uncertainty and Smoothness","date":"2023-09-29","arxiv_id":"2309.16973","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-robust-offline-to-online#ran","syntology_url":"https://syntology.ai/paper/2309.16973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16973"}},"official":{"repos":["battlewen/ro2o"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/consistency-models-as-a-rich-and-efficient","slug":"consistency-models-as-a-rich-and-efficient","title":"Consistency Models as a Rich and Efficient Policy Class for Reinforcement Learning","date":"2023-09-29","arxiv_id":"2309.16984","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":1,"n_ran_checked":6,"n_instrument":3,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"9 ran (of which 1 constructed an object rather than computing a result; 6 with no instrument failure: 3 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/consistency-models-as-a-rich-and-efficient#ran","syntology_url":"https://syntology.ai/paper/2309.16984","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16984"}},"official":{"repos":["quantumiracle/consistency_model_for_reinforcement_learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["community","official","unlocated"]}}},{"url":"/paper/conservative-world-models","slug":"conservative-world-models","title":"Zero-Shot Reinforcement Learning from Low Quality Data","date":"2023-09-26","arxiv_id":"2309.15178","repositories_listed":2,"syntology":{"n":18,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":8,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/conservative-world-models#ran","syntology_url":"https://syntology.ai/paper/2309.15178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.15178"}},"official":{"repos":["enjeeneer/conservative-world-models","enjeeneer/zero-shot-rl"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/counterfactual-conservative-q-learning-for-1","slug":"counterfactual-conservative-q-learning-for-1","title":"Counterfactual Conservative Q Learning for Offline Multi-agent Reinforcement Learning","date":"2023-09-22","arxiv_id":"2309.12696","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":3,"n_ran_checked":4,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":9,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/counterfactual-conservative-q-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/2309.12696","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.12696"}},"official":{"repos":["thu-rllab/CFCQL"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/reasoning-with-latent-diffusion-in-offline","slug":"reasoning-with-latent-diffusion-in-offline","title":"Reasoning with Latent Diffusion in Offline Reinforcement Learning","date":"2023-09-12","arxiv_id":"2309.06599","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/reasoning-with-latent-diffusion-in-offline#ran","syntology_url":"https://syntology.ai/paper/2309.06599","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.06599"}},"official":{"repos":["ldcq/ldcq"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/alphastar-unplugged-large-scale-offline","slug":"alphastar-unplugged-large-scale-offline","title":"AlphaStar Unplugged: Large-Scale Offline Reinforcement Learning","date":"2023-08-07","arxiv_id":"2308.03526","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/alphastar-unplugged-large-scale-offline#ran","syntology_url":"https://syntology.ai/paper/2308.03526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.03526"}},"official":null}},{"url":"/paper/model-based-offline-reinforcement-learning-2","slug":"model-based-offline-reinforcement-learning-2","title":"Model-based Offline Reinforcement Learning with Count-based Conservatism","date":"2023-07-21","arxiv_id":"2307.11352","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":1,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/model-based-offline-reinforcement-learning-2#ran","syntology_url":"https://syntology.ai/paper/2307.11352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11352"}},"official":{"repos":["oh-lab/count-morl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/robotic-manipulation-datasets-for-offline","slug":"robotic-manipulation-datasets-for-offline","title":"Robotic Manipulation Datasets for Offline Compositional Reinforcement Learning","date":"2023-07-13","arxiv_id":"2307.07091","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robotic-manipulation-datasets-for-offline#ran","syntology_url":"https://syntology.ai/paper/2307.07091","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.07091"}},"official":{"repos":["lifelong-ml/offline-compositional-rl-datasets"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/alleviating-matthew-effect-of-offline","slug":"alleviating-matthew-effect-of-offline","title":"Alleviating Matthew Effect of Offline Reinforcement Learning in Interactive Recommendation","date":"2023-07-10","arxiv_id":"2307.04571","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/alleviating-matthew-effect-of-offline#ran","syntology_url":"https://syntology.ai/paper/2307.04571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.04571"}},"official":{"repos":["chongminggao/dorl-codes"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/beyond-ood-state-actions-supported-cross","slug":"beyond-ood-state-actions-supported-cross","title":"Beyond OOD State Actions: Supported Cross-Domain Offline Reinforcement Learning","date":"2023-06-22","arxiv_id":"2306.12755","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-ood-state-actions-supported-cross#ran","syntology_url":"https://syntology.ai/paper/2306.12755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.12755"}},"official":{"repos":["thuml/SPOT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-regularization-with-dataset-constraint","slug":"policy-regularization-with-dataset-constraint","title":"Policy Regularization with Dataset Constraint for Offline Reinforcement Learning","date":"2023-06-11","arxiv_id":"2306.06569","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":6,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 6 samples that ran constructed an object rather than computing a result","sample_list":"/paper/policy-regularization-with-dataset-constraint#ran","syntology_url":"https://syntology.ai/paper/2306.06569","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.06569"}},"official":{"repos":["lamda-rl/prdc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/offline-prioritized-experience-replay","slug":"offline-prioritized-experience-replay","title":"Decoupled Prioritized Resampling for Offline RL","date":"2023-06-08","arxiv_id":"2306.05412","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/offline-prioritized-experience-replay#ran","syntology_url":"https://syntology.ai/paper/2306.05412","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.05412"}},"official":{"repos":["sail-sg/oper","yueyang130/odpr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/look-beneath-the-surface-exploiting-1","slug":"look-beneath-the-surface-exploiting-1","title":"Look Beneath the Surface: Exploiting Fundamental Symmetry for Sample-Efficient Offline RL","date":"2023-06-07","arxiv_id":"2306.04220","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/look-beneath-the-surface-exploiting-1#ran","syntology_url":"https://syntology.ai/paper/2306.04220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.04220"}},"official":{"repos":["pcheng2/tsrl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/primal-attention-self-attention-through","slug":"primal-attention-self-attention-through","title":"Primal-Attention: Self-attention through Asymmetric Kernel SVD in Primal Representation","date":"2023-05-31","arxiv_id":"2305.19798","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/primal-attention-self-attention-through#ran","syntology_url":"https://syntology.ai/paper/2305.19798","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.19798"}},"official":{"repos":["yingyichen-cyy/PrimalAttention"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-diffusion-policies-for-offline-1","slug":"efficient-diffusion-policies-for-offline-1","title":"Efficient Diffusion Policies for Offline Reinforcement Learning","date":"2023-05-31","arxiv_id":"2305.20081","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-diffusion-policies-for-offline-1#ran","syntology_url":"https://syntology.ai/paper/2305.20081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.20081"}},"official":{"repos":["sail-sg/edp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/what-is-essential-for-unseen-goal","slug":"what-is-essential-for-unseen-goal","title":"What is Essential for Unseen Goal Generalization of Offline Goal-conditioned RL?","date":"2023-05-30","arxiv_id":"2305.18882","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/what-is-essential-for-unseen-goal#ran","syntology_url":"https://syntology.ai/paper/2305.18882","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18882"}},"official":{"repos":["yangrui2015/goat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/madiff-offline-multi-agent-learning-with","slug":"madiff-offline-multi-agent-learning-with","title":"MADiff: Offline Multi-agent Learning with Diffusion Models","date":"2023-05-27","arxiv_id":"2305.17330","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/madiff-offline-multi-agent-learning-with#ran","syntology_url":"https://syntology.ai/paper/2305.17330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17330"}},"official":{"repos":["zbzhu99/madiff"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-reward-offline-preference-guided","slug":"beyond-reward-offline-preference-guided","title":"Beyond Reward: Offline Preference-guided Policy Optimization","date":"2023-05-25","arxiv_id":"2305.16217","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-reward-offline-preference-guided#ran","syntology_url":"https://syntology.ai/paper/2305.16217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.16217"}},"official":{"repos":["bkkgbkjb/oppo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-language-models-with-advantage","slug":"improving-language-models-with-advantage","title":"Leftover Lunch: Advantage-based Offline Reinforcement Learning for Language Models","date":"2023-05-24","arxiv_id":"2305.14718","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/improving-language-models-with-advantage#ran","syntology_url":"https://syntology.ai/paper/2305.14718","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14718"}},"official":{"repos":["abaheti95/lol-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/collaborative-world-models-an-online-offline","slug":"collaborative-world-models-an-online-offline","title":"Making Offline RL Online: Collaborative World Models for Offline Visual Reinforcement Learning","date":"2023-05-24","arxiv_id":"2305.15260","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":7,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":13,"phrase":"10 ran (of which 7 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/collaborative-world-models-an-online-offline#ran","syntology_url":"https://syntology.ai/paper/2305.15260","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15260"}},"official":{"repos":["qiwang067/CoWorld"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":7,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/2305-14550","slug":"2305-14550","title":"When should we prefer Decision Transformers for Offline Reinforcement Learning?","date":"2023-05-23","arxiv_id":"2305.14550","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/2305-14550#ran","syntology_url":"https://syntology.ai/paper/2305.14550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14550"}},"official":{"repos":["prajjwal1/rl_paradigm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/furniturebench-reproducible-real-world","slug":"furniturebench-reproducible-real-world","title":"FurnitureBench: Reproducible Real-World Benchmark for Long-Horizon Complex Manipulation","date":"2023-05-22","arxiv_id":"2305.12821","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/furniturebench-reproducible-real-world#ran","syntology_url":"https://syntology.ai/paper/2305.12821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12821"}},"official":{"repos":["clvrai/furniture-bench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-the-minimalist-approach-to-offline","slug":"revisiting-the-minimalist-approach-to-offline","title":"Revisiting the Minimalist Approach to Offline Reinforcement Learning","date":"2023-05-16","arxiv_id":"2305.09836","repositories_listed":3,"syntology":{"n":16,"n_ran":15,"n_constructed":0,"n_ran_checked":13,"n_instrument":2,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":11,"n_pointer_only":1,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 2 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/revisiting-the-minimalist-approach-to-offline#ran","syntology_url":"https://syntology.ai/paper/2305.09836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.09836"}},"official":{"repos":["dt6a/rebrac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/towards-generalizable-reinforcement-learning","slug":"towards-generalizable-reinforcement-learning","title":"Towards Generalizable Reinforcement Learning for Trade Execution","date":"2023-05-12","arxiv_id":"2307.11685","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/towards-generalizable-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2307.11685","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11685"}},"official":null}},{"url":"/paper/explaining-rl-decisions-with-trajectories","slug":"explaining-rl-decisions-with-trajectories","title":"Explaining RL Decisions with Trajectories","date":"2023-05-06","arxiv_id":"2305.04073","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/explaining-rl-decisions-with-trajectories#ran","syntology_url":"https://syntology.ai/paper/2305.04073","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.04073"}},"official":{"repos":["shripaddeshmukh/xrl_with_trajectories"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/masked-trajectory-models-for-prediction","slug":"masked-trajectory-models-for-prediction","title":"Masked Trajectory Models for Prediction, Representation, and Control","date":"2023-05-04","arxiv_id":"2305.02968","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/masked-trajectory-models-for-prediction#ran","syntology_url":"https://syntology.ai/paper/2305.02968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.02968"}},"official":{"repos":["facebookresearch/mtm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-factored-action-spaces-for","slug":"leveraging-factored-action-spaces-for","title":"Leveraging Factored Action Spaces for Efficient Offline Reinforcement Learning in Healthcare","date":"2023-05-02","arxiv_id":"2305.01738","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leveraging-factored-action-spaces-for#ran","syntology_url":"https://syntology.ai/paper/2305.01738","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.01738"}},"official":{"repos":["mld3/offlinerl_factoredactions"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/idql-implicit-q-learning-as-an-actor-critic","slug":"idql-implicit-q-learning-as-an-actor-critic","title":"IDQL: Implicit Q-Learning as an Actor-Critic Method with Diffusion Policies","date":"2023-04-20","arxiv_id":"2304.10573","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/idql-implicit-q-learning-as-an-actor-critic#ran","syntology_url":"https://syntology.ai/paper/2304.10573","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.10573"}},"official":{"repos":["philippe-eecs/idql"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/mahalo-unifying-offline-reinforcement","slug":"mahalo-unifying-offline-reinforcement","title":"MAHALO: Unifying Offline Reinforcement Learning and Imitation Learning from Observations","date":"2023-03-30","arxiv_id":"2303.17156","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mahalo-unifying-offline-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2303.17156","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17156"}},"official":{"repos":["anqili/mahalo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-rl-with-no-ood-actions-in-sample","slug":"offline-rl-with-no-ood-actions-in-sample","title":"Offline RL with No OOD Actions: In-Sample Learning via Implicit Value Regularization","date":"2023-03-28","arxiv_id":"2303.15810","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/offline-rl-with-no-ood-actions-in-sample#ran","syntology_url":"https://syntology.ai/paper/2303.15810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.15810"}},"official":{"repos":["ryanxhr/ivr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/optimal-transport-for-offline-imitation","slug":"optimal-transport-for-offline-imitation","title":"Optimal Transport for Offline Imitation Learning","date":"2023-03-24","arxiv_id":"2303.13971","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimal-transport-for-offline-imitation#ran","syntology_url":"https://syntology.ai/paper/2303.13971","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.13971"}},"official":{"repos":["ethanluoyc/optimal_transport_reward"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/decision-transformer-under-random-frame","slug":"decision-transformer-under-random-frame","title":"Decision Transformer under Random Frame Dropping","date":"2023-03-03","arxiv_id":"2303.03391","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":3,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/decision-transformer-under-random-frame#ran","syntology_url":"https://syntology.ai/paper/2303.03391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03391"}},"official":{"repos":["hukz18/defog"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-control-autonomous-fleets-from","slug":"learning-to-control-autonomous-fleets-from","title":"Learning to Control Autonomous Fleets from Observation via Offline Reinforcement Learning","date":"2023-02-28","arxiv_id":"2302.14833","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-control-autonomous-fleets-from#ran","syntology_url":"https://syntology.ai/paper/2302.14833","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.14833"}},"official":{"repos":["carolinssc/offline-rl-amod","carolinssc/offline-rl-for-amod"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/neural-laplace-control-for-continuous-time","slug":"neural-laplace-control-for-continuous-time","title":"Neural Laplace Control for Continuous-time Delayed Systems","date":"2023-02-24","arxiv_id":"2302.12604","repositories_listed":2,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":9,"n_instrument":4,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/neural-laplace-control-for-continuous-time#ran","syntology_url":"https://syntology.ai/paper/2302.12604","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12604"}},"official":{"repos":["samholt/neurallaplacecontrol","vanderschaarlab/neurallaplacecontrol"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/behavior-proximal-policy-optimization","slug":"behavior-proximal-policy-optimization","title":"Behavior Proximal Policy Optimization","date":"2023-02-22","arxiv_id":"2302.11312","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/behavior-proximal-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2302.11312","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.11312"}},"official":{"repos":["dragon-zhuang/bppo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/imitation-from-arbitrary-experience-a-dual","slug":"imitation-from-arbitrary-experience-a-dual","title":"Dual RL: Unification and New Methods for Reinforcement and Imitation Learning","date":"2023-02-16","arxiv_id":"2302.08560","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imitation-from-arbitrary-experience-a-dual#ran","syntology_url":"https://syntology.ai/paper/2302.08560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.08560"}},"official":{"repos":["hari-sikchi/DVL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/direct-preference-based-policy-optimization-1","slug":"direct-preference-based-policy-optimization-1","title":"Direct Preference-based Policy Optimization without Reward Modeling","date":"2023-01-30","arxiv_id":"2301.12842","repositories_listed":2,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/direct-preference-based-policy-optimization-1#ran","syntology_url":"https://syntology.ai/paper/2301.12842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12842"}},"official":{"repos":["snu-mllab/dppo"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/guiding-online-reinforcement-learning-with","slug":"guiding-online-reinforcement-learning-with","title":"Guiding Online Reinforcement Learning with Action-Free Offline Pretraining","date":"2023-01-30","arxiv_id":"2301.12876","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/guiding-online-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2301.12876","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12876"}},"official":{"repos":["vision-cair/af-guide"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"b0ed8ebc94d9658a6ac524188361737dd0dd3e20e3d43660a158a647f484f5da","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}