{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/offline-rl/papers/2","list_of":"/task/offline-rl","task":"Offline RL","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":8,"rows_per_page":100,"rows":[101,200],"of":755,"counts":{"archive_papers_tagged":755,"with_a_code_link":310,"where_syntology_ran_a_sample":164,"not_listed_spam_title":0,"listed":755,"listed_where_code_ran":164,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":139,"every_run_a_failure_of_syntologys_instrument":25,"listed_with_a_run_with_no_instrument_failure":139,"listed_every_run_a_failure_of_syntologys_instrument":25,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/offline-rl","prev":"/task/offline-rl","next":"/task/offline-rl/papers/3","papers":[{"url":"/paper/longreward-improving-long-context-large","slug":"longreward-improving-long-context-large","title":"LongReward: Improving Long-context Large Language Models with AI Feedback","date":"2024-10-28","arxiv_id":"2410.21252","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longreward-improving-long-context-large#ran","syntology_url":"https://syntology.ai/paper/2410.21252","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21252"}},"official":{"repos":["THUDM/LongReward"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/offline-reinforcement-learning-with-ood-state","slug":"offline-reinforcement-learning-with-ood-state","title":"Offline Reinforcement Learning with OOD State Correction and OOD Action Suppression","date":"2024-10-25","arxiv_id":"2410.19400","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/offline-reinforcement-learning-with-ood-state#ran","syntology_url":"https://syntology.ai/paper/2410.19400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.19400"}},"official":{"repos":["MAOYIXIU/SCAS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-versatile-skills-with-curriculum","slug":"learning-versatile-skills-with-curriculum","title":"Learning Versatile Skills with Curriculum Masking","date":"2024-10-23","arxiv_id":"2410.17744","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-versatile-skills-with-curriculum#ran","syntology_url":"https://syntology.ai/paper/2410.17744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17744"}},"official":{"repos":["yaotang23/currmask"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/steering-your-generalists-improving-robotic","slug":"steering-your-generalists-improving-robotic","title":"Steering Your Generalists: Improving Robotic Foundation Models via Value Guidance","date":"2024-10-17","arxiv_id":"2410.13816","repositories_listed":1,"syntology":null},{"url":"/paper/bayes-adaptive-monte-carlo-tree-search-for","slug":"bayes-adaptive-monte-carlo-tree-search-for","title":"Bayes Adaptive Monte Carlo Tree Search for Offline Model-based Reinforcement Learning","date":"2024-10-15","arxiv_id":"2410.11234","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bayes-adaptive-monte-carlo-tree-search-for#ran","syntology_url":"https://syntology.ai/paper/2410.11234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11234"}},"official":{"repos":["lucascjysdl/offline-rl-kit"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-offline-model-based-rl-via-jointly","slug":"scaling-offline-model-based-rl-via-jointly","title":"Scaling Offline Model-Based RL via Jointly-Optimized World-Action Model Pretraining","date":"2024-10-01","arxiv_id":"2410.00564","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/scaling-offline-model-based-rl-via-jointly#ran","syntology_url":"https://syntology.ai/paper/2410.00564","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00564"}},"official":{"repos":["cjreinforce/jowa"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/dmc-vb-a-benchmark-for-representation","slug":"dmc-vb-a-benchmark-for-representation","title":"DMC-VB: A Benchmark for Representation Learning for Control with Visual Distractors","date":"2024-09-26","arxiv_id":"2409.18330","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dmc-vb-a-benchmark-for-representation#ran","syntology_url":"https://syntology.ai/paper/2409.18330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.18330"}},"official":{"repos":["google-deepmind/dmc_vision_benchmark"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-role-of-deep-learning-regularizations-on","slug":"the-role-of-deep-learning-regularizations-on","title":"The Role of Deep Learning Regularizations on Actors in Offline RL","date":"2024-09-11","arxiv_id":"2409.07606","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-sample-efficiency-and-exploration","slug":"enhancing-sample-efficiency-and-exploration","title":"Enhancing Sample Efficiency and Exploration in Reinforcement Learning through the Integration of Diffusion Models and Proximal Policy Optimization","date":"2024-09-02","arxiv_id":"2409.01427","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-unlabeled-data-sharing-through","slug":"leveraging-unlabeled-data-sharing-through","title":"Leveraging Unlabeled Data Sharing through Kernel Function Approximation in Offline Reinforcement Learning","date":"2024-08-22","arxiv_id":"2408.12307","repositories_listed":1,"syntology":null},{"url":"/paper/preference-guided-reflective-sampling-for","slug":"preference-guided-reflective-sampling-for","title":"Preference-Guided Reflective Sampling for Aligning Language Models","date":"2024-08-22","arxiv_id":"2408.12163","repositories_listed":1,"syntology":null},{"url":"/paper/hokoff-real-game-dataset-from-honor-of-kings-1","slug":"hokoff-real-game-dataset-from-honor-of-kings-1","title":"Hokoff: Real Game Dataset from Honor of Kings and its Offline Reinforcement Learning Benchmarks","date":"2024-08-20","arxiv_id":"2408.10556","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hokoff-real-game-dataset-from-honor-of-kings-1#ran","syntology_url":"https://syntology.ai/paper/2408.10556","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.10556"}},"official":{"repos":["tencent-ailab/hokoff"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/experimental-evaluation-of-offline","slug":"experimental-evaluation-of-offline","title":"Experimental evaluation of offline reinforcement learning for HVAC control in buildings","date":"2024-08-15","arxiv_id":"2408.07986","repositories_listed":1,"syntology":null},{"url":"/paper/a-simulation-benchmark-for-autonomous-racing","slug":"a-simulation-benchmark-for-autonomous-racing","title":"A Simulation Benchmark for Autonomous Racing with Large-Scale Human Data","date":"2024-07-23","arxiv_id":"2407.16680","repositories_listed":1,"syntology":null},{"url":"/paper/diffusion-models-as-optimizers-for-efficient","slug":"diffusion-models-as-optimizers-for-efficient","title":"Diffusion Models as Optimizers for Efficient Planning in Offline RL","date":"2024-07-23","arxiv_id":"2407.16142","repositories_listed":1,"syntology":null},{"url":"/paper/roler-effective-reward-shaping-in-offline","slug":"roler-effective-reward-shaping-in-offline","title":"ROLeR: Effective Reward Shaping in Offline Reinforcement Learning for Recommender Systems","date":"2024-07-18","arxiv_id":"2407.13163","repositories_listed":1,"syntology":null},{"url":"/paper/digirl-training-in-the-wild-device-control","slug":"digirl-training-in-the-wild-device-control","title":"DigiRL: Training In-The-Wild Device-Control Agents with Autonomous Reinforcement Learning","date":"2024-06-14","arxiv_id":"2406.11896","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/digirl-training-in-the-wild-device-control#ran","syntology_url":"https://syntology.ai/paper/2406.11896","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11896"}},"official":null}},{"url":"/paper/is-value-learning-really-the-main-bottleneck","slug":"is-value-learning-really-the-main-bottleneck","title":"Is Value Learning Really the Main Bottleneck in Offline RL?","date":"2024-06-13","arxiv_id":"2406.09329","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":6,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/is-value-learning-really-the-main-bottleneck#ran","syntology_url":"https://syntology.ai/paper/2406.09329","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09329"}},"official":null}},{"url":"/paper/is-value-functions-estimation-with","slug":"is-value-functions-estimation-with","title":"Is Value Functions Estimation with Classification Plug-and-play for Offline Reinforcement Learning?","date":"2024-06-10","arxiv_id":"2406.06309","repositories_listed":1,"syntology":null},{"url":"/paper/plandq-hierarchical-plan-orchestration-via-d","slug":"plandq-hierarchical-plan-orchestration-via-d","title":"PlanDQ: Hierarchical Plan Orchestration via D-Conductor and Q-Performer","date":"2024-06-10","arxiv_id":"2406.06793","repositories_listed":1,"syntology":null},{"url":"/paper/decision-mamba-a-multi-grained-state-space","slug":"decision-mamba-a-multi-grained-state-space","title":"Decision Mamba: A Multi-Grained State Space Model with Self-Evolution Regularization for Offline RL","date":"2024-06-08","arxiv_id":"2406.05427","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":4,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/decision-mamba-a-multi-grained-state-space#ran","syntology_url":"https://syntology.ai/paper/2406.05427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05427"}},"official":{"repos":["aopolin-lv/DecisionMamba"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"url":"/paper/stabilizing-extreme-q-learning-by-maclaurin","slug":"stabilizing-extreme-q-learning-by-maclaurin","title":"Stabilizing Extreme Q-learning by Maclaurin Expansion","date":"2024-06-07","arxiv_id":"2406.04896","repositories_listed":1,"syntology":null},{"url":"/paper/strategically-conservative-q-learning","slug":"strategically-conservative-q-learning","title":"Strategically Conservative Q-Learning","date":"2024-06-06","arxiv_id":"2406.04534","repositories_listed":1,"syntology":null},{"url":"/paper/diffusion-policies-creating-a-trust-region","slug":"diffusion-policies-creating-a-trust-region","title":"Diffusion Policies creating a Trust Region for Offline Reinforcement Learning","date":"2024-05-30","arxiv_id":"2405.19690","repositories_listed":1,"syntology":{"n":20,"n_ran":17,"n_constructed":0,"n_ran_checked":11,"n_instrument":6,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":20,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/diffusion-policies-creating-a-trust-region#ran","syntology_url":"https://syntology.ai/paper/2405.19690","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19690"}},"official":{"repos":["tianyucodings/diffusion_trusted_q_learning"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/aligniql-policy-alignment-in-implicit-q","slug":"aligniql-policy-alignment-in-implicit-q","title":"AlignIQL: Policy Alignment in Implicit Q-Learning through Constrained Optimization","date":"2024-05-28","arxiv_id":"2405.18187","repositories_listed":1,"syntology":null},{"url":"/paper/offline-boosted-actor-critic-adaptively","slug":"offline-boosted-actor-critic-adaptively","title":"Offline-Boosted Actor-Critic: Adaptively Blending Optimal Historical Behaviors in Deep Off-Policy RL","date":"2024-05-28","arxiv_id":"2405.18520","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-in-dynamic-treatment","slug":"reinforcement-learning-in-dynamic-treatment","title":"Reinforcement Learning in Dynamic Treatment Regimes Needs Critical Reexamination","date":"2024-05-28","arxiv_id":"2405.18556","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reinforcement-learning-in-dynamic-treatment#ran","syntology_url":"https://syntology.ai/paper/2405.18556","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.18556"}},"official":{"repos":["gilesluo/reassessdtr"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/any-step-dynamics-model-improves-future","slug":"any-step-dynamics-model-improves-future","title":"Any-step Dynamics Model Improves Future Predictions for Online and Offline Reinforcement Learning","date":"2024-05-27","arxiv_id":"2405.17031","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/any-step-dynamics-model-improves-future#ran","syntology_url":"https://syntology.ai/paper/2405.17031","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17031"}},"official":{"repos":["HxLyn3/ADMPO"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gta-generative-trajectory-augmentation-with","slug":"gta-generative-trajectory-augmentation-with","title":"GTA: Generative Trajectory Augmentation with Guidance for Offline Reinforcement Learning","date":"2024-05-27","arxiv_id":"2405.16907","repositories_listed":1,"syntology":{"n":37,"n_ran":34,"n_constructed":0,"n_ran_checked":29,"n_instrument":5,"n_unverified":3,"n_honours":3,"n_violates":4,"n_no_contract":22,"n_pointer_only":5,"phrase":"34 ran (of which 0 constructed an object rather than computing a result; 29 with no instrument failure: 3 honoured, 4 violated, 22 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/gta-generative-trajectory-augmentation-with#ran","syntology_url":"https://syntology.ai/paper/2405.16907","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16907"}},"official":{"repos":["jaewoopudding/gta"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":3,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/diffusion-based-reinforcement-learning-via-q","slug":"diffusion-based-reinforcement-learning-via-q","title":"Diffusion-based Reinforcement Learning via Q-weighted Variational Policy Optimization","date":"2024-05-25","arxiv_id":"2405.16173","repositories_listed":1,"syntology":null},{"url":"/paper/generating-code-world-models-with-large","slug":"generating-code-world-models-with-large","title":"Generating Code World Models with Large Language Models Guided by Monte Carlo Tree Search","date":"2024-05-24","arxiv_id":"2405.15383","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generating-code-world-models-with-large#ran","syntology_url":"https://syntology.ai/paper/2405.15383","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15383"}},"official":{"repos":["nicoladainese96/code-world-models"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-reinforcement-learning-from-datasets","slug":"offline-reinforcement-learning-from-datasets","title":"Offline Reinforcement Learning from Datasets with Structured Non-Stationarity","date":"2024-05-23","arxiv_id":"2405.14114","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/offline-reinforcement-learning-from-datasets#ran","syntology_url":"https://syntology.ai/paper/2405.14114","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14114"}},"official":{"repos":["johannesack/offlinerlstructurednonstationarity"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/is-mamba-compatible-with-trajectory","slug":"is-mamba-compatible-with-trajectory","title":"Is Mamba Compatible with Trajectory Optimization in Offline Reinforcement Learning?","date":"2024-05-20","arxiv_id":"2405.12094","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/is-mamba-compatible-with-trajectory#ran","syntology_url":"https://syntology.ai/paper/2405.12094","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.12094"}},"official":{"repos":["AndssY/DeMa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/reinformer-max-return-sequence-modeling-for","slug":"reinformer-max-return-sequence-modeling-for","title":"Reinformer: Max-Return Sequence Modeling for Offline RL","date":"2024-05-14","arxiv_id":"2405.08740","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/reinformer-max-return-sequence-modeling-for#ran","syntology_url":"https://syntology.ai/paper/2405.08740","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.08740"}},"official":{"repos":["dragon-zhuang/reinformer"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ltldog-satisfying-temporally-extended","slug":"ltldog-satisfying-temporally-extended","title":"LTLDoG: Satisfying Temporally-Extended Symbolic Constraints for Safe Diffusion-based Planning","date":"2024-05-07","arxiv_id":"2405.04235","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ltldog-satisfying-temporally-extended#ran","syntology_url":"https://syntology.ai/paper/2405.04235","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.04235"}},"official":{"repos":["clear-nus/ltldog"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pessimistic-value-iteration-for-multi-task","slug":"pessimistic-value-iteration-for-multi-task","title":"Pessimistic Value Iteration for Multi-Task Data Sharing in Offline Reinforcement Learning","date":"2024-04-30","arxiv_id":"2404.19346","repositories_listed":1,"syntology":null},{"url":"/paper/trajdeleter-enabling-trajectory-forgetting-in","slug":"trajdeleter-enabling-trajectory-forgetting-in","title":"TrajDeleter: Enabling Trajectory Forgetting in Offline Reinforcement Learning Agents","date":"2024-04-18","arxiv_id":"2404.12530","repositories_listed":1,"syntology":null},{"url":"/paper/compositional-conservatism-a-transductive","slug":"compositional-conservatism-a-transductive","title":"Compositional Conservatism: A Transductive Approach in Offline Reinforcement Learning","date":"2024-04-06","arxiv_id":"2404.04682","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":2,"n_ran_checked":2,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/compositional-conservatism-a-transductive#ran","syntology_url":"https://syntology.ai/paper/2404.04682","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04682"}},"official":{"repos":["runamu/compositional-conservatism"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-zero-shot-reinforcement-learning","slug":"unsupervised-zero-shot-reinforcement-learning","title":"Unsupervised Zero-Shot Reinforcement Learning via Functional Reward Encodings","date":"2024-02-27","arxiv_id":"2402.17135","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":1,"n_no_contract":6,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/unsupervised-zero-shot-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2402.17135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17135"}},"official":{"repos":["kvfrans/fre"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-generative-models-for-offline-policy","slug":"deep-generative-models-for-offline-policy","title":"Deep Generative Models for Offline Policy Learning: Tutorial, Survey, and Perspectives on Future Directions","date":"2024-02-21","arxiv_id":"2402.13777","repositories_listed":1,"syntology":null},{"url":"/paper/more-3s-multimodal-based-offline","slug":"more-3s-multimodal-based-offline","title":"MORE-3S:Multimodal-based Offline Reinforcement Learning with Shared Semantic Spaces","date":"2024-02-20","arxiv_id":"2402.12845","repositories_listed":1,"syntology":null},{"url":"/paper/stitching-sub-trajectories-with-conditional","slug":"stitching-sub-trajectories-with-conditional","title":"Stitching Sub-Trajectories with Conditional Diffusion Model for Goal-Conditioned Offline RL","date":"2024-02-11","arxiv_id":"2402.07226","repositories_listed":1,"syntology":{"n":29,"n_ran":20,"n_constructed":0,"n_ran_checked":14,"n_instrument":6,"n_unverified":9,"n_honours":2,"n_violates":0,"n_no_contract":12,"n_pointer_only":29,"phrase":"20 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 2 honoured, 0 violated, 12 with no contract checked; 6 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/stitching-sub-trajectories-with-conditional#ran","syntology_url":"https://syntology.ai/paper/2402.07226","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07226"}},"official":{"repos":["rlatjddbs/ssd"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/entropy-regularized-diffusion-policy-with-q","slug":"entropy-regularized-diffusion-policy-with-q","title":"Entropy-regularized Diffusion Policy with Q-Ensembles for Offline Reinforcement Learning","date":"2024-02-06","arxiv_id":"2402.04080","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/entropy-regularized-diffusion-policy-with-q#ran","syntology_url":"https://syntology.ai/paper/2402.04080","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.04080"}},"official":{"repos":["ruoqizzz/entropy-offlineRL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/seabo-a-simple-search-based-method-for","slug":"seabo-a-simple-search-based-method-for","title":"SEABO: A Simple Search-Based Method for Offline Imitation Learning","date":"2024-02-06","arxiv_id":"2402.03807","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/seabo-a-simple-search-based-method-for#ran","syntology_url":"https://syntology.ai/paper/2402.03807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03807"}},"official":{"repos":["dmksjfl/seabo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-an-information-theoretic-framework-of","slug":"towards-an-information-theoretic-framework-of","title":"Towards an Information Theoretic Framework of Context-Based Offline Meta-Reinforcement Learning","date":"2024-02-04","arxiv_id":"2402.02429","repositories_listed":1,"syntology":null},{"url":"/paper/odice-revealing-the-mystery-of-distribution","slug":"odice-revealing-the-mystery-of-distribution","title":"ODICE: Revealing the Mystery of Distribution Correction Estimation via Orthogonal-gradient Update","date":"2024-02-01","arxiv_id":"2402.00348","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":1,"n_ran_checked":5,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/odice-revealing-the-mystery-of-distribution#ran","syntology_url":"https://syntology.ai/paper/2402.00348","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.00348"}},"official":{"repos":["maoliyuan/odice-pytorch"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":1,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/differentiable-tree-search-in-latent-state","slug":"differentiable-tree-search-in-latent-state","title":"Differentiable Tree Search Network","date":"2024-01-22","arxiv_id":"2401.11660","repositories_listed":1,"syntology":null},{"url":"/paper/reframing-offline-reinforcement-learning-as-a","slug":"reframing-offline-reinforcement-learning-as-a","title":"Solving Offline Reinforcement Learning with Decision Tree Regression","date":"2024-01-21","arxiv_id":"2401.11630","repositories_listed":1,"syntology":null},{"url":"/paper/safe-offline-reinforcement-learning-with","slug":"safe-offline-reinforcement-learning-with","title":"Safe Offline Reinforcement Learning with Feasibility-Guided Diffusion Model","date":"2024-01-19","arxiv_id":"2401.10700","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":2,"n_no_contract":5,"n_pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 2 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safe-offline-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2401.10700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10700"}},"official":{"repos":["zhengyinan-air/fisor"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/diffclone-enhanced-behaviour-cloning-in","slug":"diffclone-enhanced-behaviour-cloning-in","title":"DiffClone: Enhanced Behaviour Cloning in Robotics with Diffusion-Driven Policy Learning","date":"2024-01-17","arxiv_id":"2401.09243","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diffclone-enhanced-behaviour-cloning-in#ran","syntology_url":"https://syntology.ai/paper/2401.09243","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09243"}},"official":{"repos":["sirabas369/diffclone"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-from-sparse-offline-datasets-via","slug":"learning-from-sparse-offline-datasets-via","title":"Learning from Sparse Offline Datasets via Conservative Density Estimation","date":"2024-01-16","arxiv_id":"2401.08819","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-from-sparse-offline-datasets-via#ran","syntology_url":"https://syntology.ai/paper/2401.08819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.08819"}},"official":{"repos":["czp16/cde-offline-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/spqr-controlling-q-ensemble-independence-with-1","slug":"spqr-controlling-q-ensemble-independence-with-1","title":"SPQR: Controlling Q-ensemble Independence with Spiked Random Model for Reinforcement Learning","date":"2024-01-06","arxiv_id":"2401.03137","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/spqr-controlling-q-ensemble-independence-with-1#ran","syntology_url":"https://syntology.ai/paper/2401.03137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.03137"}},"official":{"repos":["dohyeoklee/SPQR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-regularized-offline-multi-objective","slug":"policy-regularized-offline-multi-objective","title":"Policy-regularized Offline Multi-objective Reinforcement Learning","date":"2024-01-04","arxiv_id":"2401.02244","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/policy-regularized-offline-multi-objective#ran","syntology_url":"https://syntology.ai/paper/2401.02244","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.02244"}},"official":{"repos":["qianlin04/prmorl"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/poce-primal-policy-optimization-with","slug":"poce-primal-policy-optimization-with","title":"POCE: Primal Policy Optimization with Conservative Estimation for Multi-constraint Offline Reinforcement Learning","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/online-symbolic-music-alignment-with-offline","slug":"online-symbolic-music-alignment-with-offline","title":"Online Symbolic Music Alignment with Offline Reinforcement Learning","date":"2023-12-31","arxiv_id":"2401.00466","repositories_listed":1,"syntology":null},{"url":"/paper/critic-guided-decision-transformer-for","slug":"critic-guided-decision-transformer-for","title":"Critic-Guided Decision Transformer for Offline Reinforcement Learning","date":"2023-12-21","arxiv_id":"2312.13716","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/critic-guided-decision-transformer-for#ran","syntology_url":"https://syntology.ai/paper/2312.13716","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.13716"}},"official":{"repos":["sharkwyf/cgdt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/a-perspective-of-q-value-estimation-on","slug":"a-perspective-of-q-value-estimation-on","title":"A Perspective of Q-value Estimation on Offline-to-Online Reinforcement Learning","date":"2023-12-12","arxiv_id":"2312.07685","repositories_listed":1,"syntology":null},{"url":"/paper/traffic-signal-control-using-lightweight","slug":"traffic-signal-control-using-lightweight","title":"Traffic Signal Control Using Lightweight Transformers: An Offline-to-Online RL Approach","date":"2023-12-12","arxiv_id":"2312.07795","repositories_listed":1,"syntology":null},{"url":"/paper/the-generalization-gap-in-offline","slug":"the-generalization-gap-in-offline","title":"The Generalization Gap in Offline Reinforcement Learning","date":"2023-12-10","arxiv_id":"2312.05742","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/the-generalization-gap-in-offline#ran","syntology_url":"https://syntology.ai/paper/2312.05742","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.05742"}},"official":{"repos":["facebookresearch/gen_dgrl"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/micro-model-based-offline-reinforcement","slug":"micro-model-based-offline-reinforcement","title":"MICRO: Model-Based Offline Reinforcement Learning with a Conservative Bellman Operator","date":"2023-12-07","arxiv_id":"2312.03991","repositories_listed":1,"syntology":null},{"url":"/paper/scope-rl-a-python-library-for-offline","slug":"scope-rl-a-python-library-for-offline","title":"SCOPE-RL: A Python Library for Offline Reinforcement Learning and Off-Policy Evaluation","date":"2023-11-30","arxiv_id":"2311.18206","repositories_listed":1,"syntology":null},{"url":"/paper/offline-data-enhanced-on-policy-policy","slug":"offline-data-enhanced-on-policy-policy","title":"Offline Data Enhanced On-Policy Policy Gradient with Provable Guarantees","date":"2023-11-14","arxiv_id":"2311.08384","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/offline-data-enhanced-on-policy-policy#ran","syntology_url":"https://syntology.ai/paper/2311.08384","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08384"}},"official":{"repos":["yifeizhou02/hnpg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/unleashing-the-power-of-pre-trained-language","slug":"unleashing-the-power-of-pre-trained-language","title":"Unleashing the Power of Pre-trained Language Models for Offline Reinforcement Learning","date":"2023-10-31","arxiv_id":"2310.20587","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unleashing-the-power-of-pre-trained-language#ran","syntology_url":"https://syntology.ai/paper/2310.20587","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20587"}},"official":{"repos":["srzer/LaMo-2023"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/free-from-bellman-completeness-trajectory","slug":"free-from-bellman-completeness-trajectory","title":"Free from Bellman Completeness: Trajectory Stitching via Model-based Return-conditioned Supervised Learning","date":"2023-10-30","arxiv_id":"2310.19308","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/free-from-bellman-completeness-trajectory#ran","syntology_url":"https://syntology.ai/paper/2310.19308","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19308"}},"official":{"repos":["zhaoyizhou1123/mbrcsl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/robust-offline-policy-evaluation-and","slug":"robust-offline-policy-evaluation-and","title":"Robust Offline Reinforcement learning with Heavy-Tailed Rewards","date":"2023-10-28","arxiv_id":"2310.18715","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-offline-policy-evaluation-and#ran","syntology_url":"https://syntology.ai/paper/2310.18715","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.18715"}},"official":{"repos":["mamba413/room"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-distributionally-robust-learning-and","slug":"bridging-distributionally-robust-learning-and","title":"Bridging Distributionally Robust Learning and Offline RL: An Approach to Mitigate Distribution Shift and Partial Data Coverage","date":"2023-10-27","arxiv_id":"2310.18434","repositories_listed":1,"syntology":null},{"url":"/paper/crop-conservative-reward-for-model-based","slug":"crop-conservative-reward-for-model-based","title":"CROP: Conservative Reward for Model-based Offline Policy Optimization","date":"2023-10-26","arxiv_id":"2310.17245","repositories_listed":1,"syntology":null},{"url":"/paper/corruption-robust-offline-reinforcement-2","slug":"corruption-robust-offline-reinforcement-2","title":"Corruption-Robust Offline Reinforcement Learning with General Function Approximation","date":"2023-10-23","arxiv_id":"2310.14550","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/corruption-robust-offline-reinforcement-2#ran","syntology_url":"https://syntology.ai/paper/2310.14550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14550"}},"official":{"repos":["yangrui2015/uwmsg"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/building-persona-consistent-dialogue-agents","slug":"building-persona-consistent-dialogue-agents","title":"Building Persona Consistent Dialogue Agents with Offline Reinforcement Learning","date":"2023-10-16","arxiv_id":"2310.10735","repositories_listed":1,"syntology":null},{"url":"/paper/offline-retraining-for-online-rl-decoupled","slug":"offline-retraining-for-online-rl-decoupled","title":"Offline Retraining for Online RL: Decoupled Policy Learning to Mitigate Exploration Bias","date":"2023-10-12","arxiv_id":"2310.08558","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/offline-retraining-for-online-rl-decoupled#ran","syntology_url":"https://syntology.ai/paper/2310.08558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08558"}},"official":{"repos":["MaxSobolMark/OOO"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/diffcps-diffusion-model-based-constrained","slug":"diffcps-diffusion-model-based-constrained","title":"DiffCPS: Diffusion Model based Constrained Policy Search for Offline Reinforcement Learning","date":"2023-10-09","arxiv_id":"2310.05333","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":4,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"9 ran (of which 4 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/diffcps-diffusion-model-based-constrained#ran","syntology_url":"https://syntology.ai/paper/2310.05333","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05333"}},"official":{"repos":["felix-thu/DiffCPS"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":4,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-uniform-sampling-offline-reinforcement-1","slug":"beyond-uniform-sampling-offline-reinforcement-1","title":"Beyond Uniform Sampling: Offline Reinforcement Learning with Imbalanced Datasets","date":"2023-10-06","arxiv_id":"2310.04413","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/beyond-uniform-sampling-offline-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2310.04413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04413"}},"official":{"repos":["Improbable-AI/dw-offline-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-reach-goals-via-diffusion","slug":"learning-to-reach-goals-via-diffusion","title":"Learning to Reach Goals via Diffusion","date":"2023-10-04","arxiv_id":"2310.02505","repositories_listed":1,"syntology":null},{"url":"/paper/consistency-models-as-a-rich-and-efficient","slug":"consistency-models-as-a-rich-and-efficient","title":"Consistency Models as a Rich and Efficient Policy Class for Reinforcement Learning","date":"2023-09-29","arxiv_id":"2309.16984","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":1,"n_ran_checked":6,"n_instrument":3,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"9 ran (of which 1 constructed an object rather than computing a result; 6 with no instrument failure: 3 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/consistency-models-as-a-rich-and-efficient#ran","syntology_url":"https://syntology.ai/paper/2309.16984","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16984"}},"official":{"repos":["quantumiracle/consistency_model_for_reinforcement_learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["community","official","unlocated"]}}},{"url":"/paper/towards-robust-offline-to-online","slug":"towards-robust-offline-to-online","title":"Towards Robust Offline-to-Online Reinforcement Learning via Uncertainty and Smoothness","date":"2023-09-29","arxiv_id":"2309.16973","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-robust-offline-to-online#ran","syntology_url":"https://syntology.ai/paper/2309.16973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16973"}},"official":{"repos":["battlewen/ro2o"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/counterfactual-conservative-q-learning-for-1","slug":"counterfactual-conservative-q-learning-for-1","title":"Counterfactual Conservative Q Learning for Offline Multi-agent Reinforcement Learning","date":"2023-09-22","arxiv_id":"2309.12696","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":3,"n_ran_checked":4,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":9,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/counterfactual-conservative-q-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/2309.12696","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.12696"}},"official":{"repos":["thu-rllab/CFCQL"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/vapor-holonomic-legged-robot-navigation-in","slug":"vapor-holonomic-legged-robot-navigation-in","title":"VAPOR: Legged Robot Navigation in Outdoor Vegetation Using Offline Reinforcement Learning","date":"2023-09-14","arxiv_id":"2309.07832","repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-with-latent-diffusion-in-offline","slug":"reasoning-with-latent-diffusion-in-offline","title":"Reasoning with Latent Diffusion in Offline Reinforcement Learning","date":"2023-09-12","arxiv_id":"2309.06599","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/reasoning-with-latent-diffusion-in-offline#ran","syntology_url":"https://syntology.ai/paper/2309.06599","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.06599"}},"official":{"repos":["ldcq/ldcq"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/orl-auditor-dataset-auditing-in-offline-deep","slug":"orl-auditor-dataset-auditing-in-offline-deep","title":"ORL-AUDITOR: Dataset Auditing in Offline Deep Reinforcement Learning","date":"2023-09-06","arxiv_id":"2309.03081","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-offline-policy-optimization-with","slug":"model-based-offline-policy-optimization-with","title":"Model-based Offline Policy Optimization with Adversarial Network","date":"2023-09-05","arxiv_id":"2309.02157","repositories_listed":1,"syntology":null},{"url":"/paper/alphastar-unplugged-large-scale-offline","slug":"alphastar-unplugged-large-scale-offline","title":"AlphaStar Unplugged: Large-Scale Offline Reinforcement Learning","date":"2023-08-07","arxiv_id":"2308.03526","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/alphastar-unplugged-large-scale-offline#ran","syntology_url":"https://syntology.ai/paper/2308.03526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.03526"}},"official":null}},{"url":"/paper/a-connection-between-one-step-regularization","slug":"a-connection-between-one-step-regularization","title":"A Connection between One-Step Regularization and Critic Regularization in Reinforcement Learning","date":"2023-07-24","arxiv_id":"2307.12968","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-example-based-control","slug":"contrastive-example-based-control","title":"Contrastive Example-Based Control","date":"2023-07-24","arxiv_id":"2307.13101","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-effectiveness-of-offline-rl-for","slug":"on-the-effectiveness-of-offline-rl-for","title":"On the Effectiveness of Offline RL for Dialogue Response Generation","date":"2023-07-23","arxiv_id":"2307.12425","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/on-the-effectiveness-of-offline-rl-for#ran","syntology_url":"https://syntology.ai/paper/2307.12425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12425"}},"official":{"repos":["asappresearch/dialogue-offline-rl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/model-based-offline-reinforcement-learning-2","slug":"model-based-offline-reinforcement-learning-2","title":"Model-based Offline Reinforcement Learning with Count-based Conservatism","date":"2023-07-21","arxiv_id":"2307.11352","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":1,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/model-based-offline-reinforcement-learning-2#ran","syntology_url":"https://syntology.ai/paper/2307.11352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11352"}},"official":{"repos":["oh-lab/count-morl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-self-assembling-artificial-neural","slug":"towards-self-assembling-artificial-neural","title":"Towards Self-Assembling Artificial Neural Networks through Neural Developmental Programs","date":"2023-07-17","arxiv_id":"2307.08197","repositories_listed":1,"syntology":null},{"url":"/paper/robotic-manipulation-datasets-for-offline","slug":"robotic-manipulation-datasets-for-offline","title":"Robotic Manipulation Datasets for Offline Compositional Reinforcement Learning","date":"2023-07-13","arxiv_id":"2307.07091","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robotic-manipulation-datasets-for-offline#ran","syntology_url":"https://syntology.ai/paper/2307.07091","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.07091"}},"official":{"repos":["lifelong-ml/offline-compositional-rl-datasets"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-ood-state-actions-supported-cross","slug":"beyond-ood-state-actions-supported-cross","title":"Beyond OOD State Actions: Supported Cross-Domain Offline Reinforcement Learning","date":"2023-06-22","arxiv_id":"2306.12755","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-ood-state-actions-supported-cross#ran","syntology_url":"https://syntology.ai/paper/2306.12755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.12755"}},"official":{"repos":["thuml/SPOT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/harnessing-mixed-offline-reinforcement","slug":"harnessing-mixed-offline-reinforcement","title":"Harnessing Mixed Offline Reinforcement Learning Datasets via Trajectory Weighting","date":"2023-06-22","arxiv_id":"2306.13085","repositories_listed":1,"syntology":null},{"url":"/paper/semi-offline-reinforcement-learning-for","slug":"semi-offline-reinforcement-learning-for","title":"Semi-Offline Reinforcement Learning for Optimized Text Generation","date":"2023-06-16","arxiv_id":"2306.09712","repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-evaluation-in-doubly-inhomogeneous","slug":"off-policy-evaluation-in-doubly-inhomogeneous","title":"Off-policy Evaluation in Doubly Inhomogeneous Environments","date":"2023-06-14","arxiv_id":"2306.08719","repositories_listed":1,"syntology":null},{"url":"/paper/look-beneath-the-surface-exploiting-1","slug":"look-beneath-the-surface-exploiting-1","title":"Look Beneath the Surface: Exploiting Fundamental Symmetry for Sample-Efficient Offline RL","date":"2023-06-07","arxiv_id":"2306.04220","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/look-beneath-the-surface-exploiting-1#ran","syntology_url":"https://syntology.ai/paper/2306.04220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.04220"}},"official":{"repos":["pcheng2/tsrl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mildly-constrained-evaluation-policy-for","slug":"mildly-constrained-evaluation-policy-for","title":"Mildly Constrained Evaluation Policy for Offline Reinforcement Learning","date":"2023-06-06","arxiv_id":"2306.03680","repositories_listed":1,"syntology":null},{"url":"/paper/improving-and-benchmarking-offline","slug":"improving-and-benchmarking-offline","title":"Improving and Benchmarking Offline Reinforcement Learning Algorithms","date":"2023-06-01","arxiv_id":"2306.00972","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-diffusion-policies-for-offline-1","slug":"efficient-diffusion-policies-for-offline-1","title":"Efficient Diffusion Policies for Offline Reinforcement Learning","date":"2023-05-31","arxiv_id":"2305.20081","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-diffusion-policies-for-offline-1#ran","syntology_url":"https://syntology.ai/paper/2305.20081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.20081"}},"official":{"repos":["sail-sg/edp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/primal-attention-self-attention-through","slug":"primal-attention-self-attention-through","title":"Primal-Attention: Self-attention through Asymmetric Kernel SVD in Primal Representation","date":"2023-05-31","arxiv_id":"2305.19798","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/primal-attention-self-attention-through#ran","syntology_url":"https://syntology.ai/paper/2305.19798","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.19798"}},"official":{"repos":["yingyichen-cyy/PrimalAttention"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-reinforcement-learning-objectives-for","slug":"robust-reinforcement-learning-objectives-for","title":"Robust Reinforcement Learning Objectives for Sequential Recommender Systems","date":"2023-05-30","arxiv_id":"2305.18820","repositories_listed":1,"syntology":null},{"url":"/paper/what-is-essential-for-unseen-goal","slug":"what-is-essential-for-unseen-goal","title":"What is Essential for Unseen Goal Generalization of Offline Goal-conditioned RL?","date":"2023-05-30","arxiv_id":"2305.18882","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/what-is-essential-for-unseen-goal#ran","syntology_url":"https://syntology.ai/paper/2305.18882","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18882"}},"official":{"repos":["yangrui2015/goat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/madiff-offline-multi-agent-learning-with","slug":"madiff-offline-multi-agent-learning-with","title":"MADiff: Offline Multi-agent Learning with Diffusion Models","date":"2023-05-27","arxiv_id":"2305.17330","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/madiff-offline-multi-agent-learning-with#ran","syntology_url":"https://syntology.ai/paper/2305.17330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17330"}},"official":{"repos":["zbzhu99/madiff"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-reward-offline-preference-guided","slug":"beyond-reward-offline-preference-guided","title":"Beyond Reward: Offline Preference-guided Policy Optimization","date":"2023-05-25","arxiv_id":"2305.16217","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-reward-offline-preference-guided#ran","syntology_url":"https://syntology.ai/paper/2305.16217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.16217"}},"official":{"repos":["bkkgbkjb/oppo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"f5a813c3a6b851b3a099339a823d3ef22f2b80e1307e3089de47b7aa0103bcdf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}