{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/14","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":14,"pages_in_order":152,"rows_per_page":100,"rows":[1301,1400],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/13","next":"/task/reinforcement-learning-1/papers/15","papers":[{"url":"/paper/optimistic-critic-reconstruction-and","slug":"optimistic-critic-reconstruction-and","title":"Optimistic Critic Reconstruction and Constrained Fine-Tuning for General Offline-to-Online RL","date":"2024-12-25","arxiv_id":"2412.18855","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":3,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/optimistic-critic-reconstruction-and#ran","syntology_url":"https://syntology.ai/paper/2412.18855","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.18855"}},"official":{"repos":["QinwenLuo/OCR-CFT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/autonomous-option-invention-for-continual","slug":"autonomous-option-invention-for-continual","title":"Autonomous Option Invention for Continual Hierarchical Reinforcement Learning and Planning","date":"2024-12-20","arxiv_id":"2412.16395","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-time-scale","slug":"deep-reinforcement-learning-with-time-scale","title":"Deep reinforcement learning with time-scale invariant memory","date":"2024-12-19","arxiv_id":"2412.15292","repositories_listed":1,"syntology":null},{"url":"/paper/offline-safe-reinforcement-learning-using","slug":"offline-safe-reinforcement-learning-using","title":"Offline Safe Reinforcement Learning Using Trajectory Classification","date":"2024-12-19","arxiv_id":"2412.15429","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/offline-safe-reinforcement-learning-using#ran","syntology_url":"https://syntology.ai/paper/2412.15429","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.15429"}},"official":{"repos":["zgong11/TraC"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enabling-realtime-reinforcement-learning-at","slug":"enabling-realtime-reinforcement-learning-at","title":"Enabling Realtime Reinforcement Learning at Scale with Staggered Asynchronous Inference","date":"2024-12-18","arxiv_id":"2412.14355","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enabling-realtime-reinforcement-learning-at#ran","syntology_url":"https://syntology.ai/paper/2412.14355","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.14355"}},"official":{"repos":["cerc-aai/realtime_rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/guiding-generative-protein-language-models","slug":"guiding-generative-protein-language-models","title":"Guiding Generative Protein Language Models with Reinforcement Learning","date":"2024-12-17","arxiv_id":"2412.12979","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/guiding-generative-protein-language-models#ran","syntology_url":"https://syntology.ai/paper/2412.12979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.12979"}},"official":{"repos":["ai4pdlab/dpo_plm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tilted-quantile-gradient-updates-for-quantile","slug":"tilted-quantile-gradient-updates-for-quantile","title":"Tilted Quantile Gradient Updates for Quantile-Constrained Reinforcement Learning","date":"2024-12-17","arxiv_id":"2412.13184","repositories_listed":1,"syntology":null},{"url":"/paper/rl-llm-dt-an-automatic-decision-tree","slug":"rl-llm-dt-an-automatic-decision-tree","title":"RL-LLM-DT: An Automatic Decision Tree Generation Method Based on RL Evaluation and LLM Enhancement","date":"2024-12-16","arxiv_id":"2412.11417","repositories_listed":1,"syntology":null},{"url":"/paper/using-machine-learning-to-inform-harvest","slug":"using-machine-learning-to-inform-harvest","title":"Using machine learning to inform harvest control rule design in complex fishery settings","date":"2024-12-16","arxiv_id":"2412.12400","repositories_listed":1,"syntology":null},{"url":"/paper/are-expressive-models-truly-necessary-for","slug":"are-expressive-models-truly-necessary-for","title":"Are Expressive Models Truly Necessary for Offline RL?","date":"2024-12-15","arxiv_id":"2412.11253","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/are-expressive-models-truly-necessary-for#ran","syntology_url":"https://syntology.ai/paper/2412.11253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.11253"}},"official":{"repos":["imoneoi/RSP_JAX"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/entropy-regularized-process-reward-model","slug":"entropy-regularized-process-reward-model","title":"Entropy-Regularized Process Reward Model","date":"2024-12-15","arxiv_id":"2412.11006","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/entropy-regularized-process-reward-model#ran","syntology_url":"https://syntology.ai/paper/2412.11006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.11006"}},"official":{"repos":["hanningzhang/er-prm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/adaptive-reward-design-for-reinforcement","slug":"adaptive-reward-design-for-reinforcement","title":"Adaptive Reward Design for Reinforcement Learning","date":"2024-12-14","arxiv_id":"2412.10917","repositories_listed":1,"syntology":null},{"url":"/paper/latent-safety-constrained-policy-approach-for","slug":"latent-safety-constrained-policy-approach-for","title":"Latent Safety-Constrained Policy Approach for Safe Offline Reinforcement Learning","date":"2024-12-11","arxiv_id":"2412.08794","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/latent-safety-constrained-policy-approach-for#ran","syntology_url":"https://syntology.ai/paper/2412.08794","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.08794"}},"official":{"repos":["PrajwalKoirala/LSPC-Safe-Offline-RL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sinergym-a-virtual-testbed-for-building","slug":"sinergym-a-virtual-testbed-for-building","title":"SINERGYM -- A virtual testbed for building energy optimization with Reinforcement Learning","date":"2024-12-11","arxiv_id":"2412.08293","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-online-reinforcement-learning-fine","slug":"efficient-online-reinforcement-learning-fine","title":"Efficient Online Reinforcement Learning Fine-Tuning Need Not Retain Offline Data","date":"2024-12-10","arxiv_id":"2412.07762","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-online-reinforcement-learning-fine#ran","syntology_url":"https://syntology.ai/paper/2412.07762","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.07762"}},"official":{"repos":["zhouzypaul/wsrl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-policy-as-macro","slug":"reinforcement-learning-policy-as-macro","title":"Reinforcement Learning Policy as Macro Regulator Rather than Macro Placer","date":"2024-12-10","arxiv_id":"2412.07167","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reinforcement-learning-policy-as-macro#ran","syntology_url":"https://syntology.ai/paper/2412.07167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.07167"}},"official":{"repos":["lamda-bbo/macro-regulator"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/maniskill-hab-a-benchmark-for-low-level","slug":"maniskill-hab-a-benchmark-for-low-level","title":"ManiSkill-HAB: A Benchmark for Low-Level Manipulation in Home Rearrangement Tasks","date":"2024-12-09","arxiv_id":"2412.13211","repositories_listed":1,"syntology":null},{"url":"/paper/m-3-pc-test-time-model-predictive-control-for","slug":"m-3-pc-test-time-model-predictive-control-for","title":"M$^3$PC: Test-time Model Predictive Control for Pretrained Masked Trajectory Model","date":"2024-12-07","arxiv_id":"2412.05675","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/m-3-pc-test-time-model-predictive-control-for#ran","syntology_url":"https://syntology.ai/paper/2412.05675","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.05675"}},"official":{"repos":["wkh923/m3pc"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/marvel-accelerating-safe-online-reinforcement","slug":"marvel-accelerating-safe-online-reinforcement","title":"Marvel: Accelerating Safe Online Reinforcement Learning with Finetuned Offline Policy","date":"2024-12-05","arxiv_id":"2412.04426","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-enhanced-llms-a-survey","slug":"reinforcement-learning-enhanced-llms-a-survey","title":"Reinforcement Learning Enhanced LLMs: A Survey","date":"2024-12-05","arxiv_id":"2412.10400","repositories_listed":1,"syntology":null},{"url":"/paper/towards-generalizable-autonomous-penetration","slug":"towards-generalizable-autonomous-penetration","title":"Mind the Gap: Towards Generalizable Autonomous Penetration Testing via Domain Randomization and Meta-Reinforcement Learning","date":"2024-12-05","arxiv_id":"2412.04078","repositories_listed":1,"syntology":null},{"url":"/paper/ai-driven-day-to-day-route-choice","slug":"ai-driven-day-to-day-route-choice","title":"AI-Driven Day-to-Day Route Choice","date":"2024-12-04","arxiv_id":"2412.03338","repositories_listed":1,"syntology":null},{"url":"/paper/learning-on-one-mode-addressing-multi","slug":"learning-on-one-mode-addressing-multi","title":"Learning on One Mode: Addressing Multi-Modality in Offline Reinforcement Learning","date":"2024-12-04","arxiv_id":"2412.03258","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/learning-on-one-mode-addressing-multi#ran","syntology_url":"https://syntology.ai/paper/2412.03258","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.03258"}},"official":{"repos":["MianchuWang/LOM"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/conformal-symplectic-optimization-for-stable","slug":"conformal-symplectic-optimization-for-stable","title":"Conformal Symplectic Optimization for Stable Reinforcement Learning","date":"2024-12-03","arxiv_id":"2412.02291","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-to-learn-quantum","slug":"reinforcement-learning-to-learn-quantum","title":"Reinforcement learning to learn quantum states for Heisenberg scaling accuracy","date":"2024-12-03","arxiv_id":"2412.02334","repositories_listed":1,"syntology":null},{"url":"/paper/approximately-optimal-search-on-a-higher","slug":"approximately-optimal-search-on-a-higher","title":"Approximately Optimal Search on a Higher-dimensional Sliding Puzzle","date":"2024-12-02","arxiv_id":"2412.01937","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-generative-policies-a-simpler","slug":"revisiting-generative-policies-a-simpler","title":"Revisiting Generative Policies: A Simpler Reinforcement Learning Algorithmic Perspective","date":"2024-12-02","arxiv_id":"2412.01245","repositories_listed":1,"syntology":null},{"url":"/paper/o1-coder-an-o1-replication-for-coding","slug":"o1-coder-an-o1-replication-for-coding","title":"o1-Coder: an o1 Replication for Coding","date":"2024-11-29","arxiv_id":"2412.00154","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/o1-coder-an-o1-replication-for-coding#ran","syntology_url":"https://syntology.ai/paper/2412.00154","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.00154"}},"official":{"repos":["adam-bjtu/o1-coder"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/supervised-learning-enhanced-multi-group","slug":"supervised-learning-enhanced-multi-group","title":"Supervised Learning-enhanced Multi-Group Actor Critic for Live Stream Allocation in Feed","date":"2024-11-28","arxiv_id":"2412.10381","repositories_listed":1,"syntology":null},{"url":"/paper/pretrained-llm-adapted-with-lora-as-a","slug":"pretrained-llm-adapted-with-lora-as-a","title":"Pretrained LLM Adapted with LoRA as a Decision Transformer for Offline RL in Quantitative Trading","date":"2024-11-26","arxiv_id":"2411.17900","repositories_listed":1,"syntology":null},{"url":"/paper/marco-o1-towards-open-reasoning-models-for","slug":"marco-o1-towards-open-reasoning-models-for","title":"Marco-o1: Towards Open Reasoning Models for Open-Ended Solutions","date":"2024-11-21","arxiv_id":"2411.14405","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/marco-o1-towards-open-reasoning-models-for#ran","syntology_url":"https://syntology.ai/paper/2411.14405","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.14405"}},"official":{"repos":["aidc-ai/marco-o1"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-environments-for-vehicle-routing","slug":"multi-agent-environments-for-vehicle-routing","title":"Multi-Agent Environments for Vehicle Routing Problems","date":"2024-11-21","arxiv_id":"2411.14411","repositories_listed":1,"syntology":null},{"url":"/paper/natural-language-reinforcement-learning-1","slug":"natural-language-reinforcement-learning-1","title":"Natural Language Reinforcement Learning","date":"2024-11-21","arxiv_id":"2411.14251","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/natural-language-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2411.14251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.14251"}},"official":{"repos":["waterhorse1/natural-language-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/umbrella-reinforcement-learning","slug":"umbrella-reinforcement-learning","title":"Umbrella Reinforcement Learning -- computationally efficient tool for hard non-linear problems","date":"2024-11-21","arxiv_id":"2411.14117","repositories_listed":1,"syntology":null},{"url":"/paper/acing-actor-critic-for-instruction-learning","slug":"acing-actor-critic-for-instruction-learning","title":"ACING: Actor-Critic for Instruction Learning in Black-Box Large Language Models","date":"2024-11-19","arxiv_id":"2411.12736","repositories_listed":1,"syntology":null},{"url":"/paper/action-attentive-deep-reinforcement-learning","slug":"action-attentive-deep-reinforcement-learning","title":"Action-Attentive Deep Reinforcement Learning for Autonomous Alignment of Beamlines","date":"2024-11-19","arxiv_id":"2411.12183","repositories_listed":1,"syntology":null},{"url":"/paper/ledro-llm-enhanced-design-space-reduction-and","slug":"ledro-llm-enhanced-design-space-reduction-and","title":"LEDRO: LLM-Enhanced Design Space Reduction and Optimization for Analog Circuits","date":"2024-11-19","arxiv_id":"2411.12930","repositories_listed":1,"syntology":null},{"url":"/paper/continual-task-learning-through-adaptive","slug":"continual-task-learning-through-adaptive","title":"Continual Task Learning through Adaptive Policy Self-Composition","date":"2024-11-18","arxiv_id":"2411.11364","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":4,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/continual-task-learning-through-adaptive#ran","syntology_url":"https://syntology.ai/paper/2411.11364","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.11364"}},"official":{"repos":["charleshsc/CompoFormer"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/amago-2-breaking-the-multi-task-barrier-in","slug":"amago-2-breaking-the-multi-task-barrier-in","title":"AMAGO-2: Breaking the Multi-Task Barrier in Meta-Reinforcement Learning with Transformers","date":"2024-11-17","arxiv_id":"2411.11188","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/amago-2-breaking-the-multi-task-barrier-in#ran","syntology_url":"https://syntology.ai/paper/2411.11188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.11188"}},"official":{"repos":["ut-austin-rpl/amago"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/an-investigation-of-offline-reinforcement","slug":"an-investigation-of-offline-reinforcement","title":"An Investigation of Offline Reinforcement Learning in Factorisable Action Spaces","date":"2024-11-17","arxiv_id":"2411.11088","repositories_listed":1,"syntology":null},{"url":"/paper/recommender-systems-and-reinforcement","slug":"recommender-systems-and-reinforcement","title":"Recommender systems and reinforcement learning for human-building interaction and context-aware support: A text mining-driven review of scientific literature","date":"2024-11-13","arxiv_id":"2411.08734","repositories_listed":1,"syntology":null},{"url":"/paper/doubly-mild-generalization-for-offline","slug":"doubly-mild-generalization-for-offline","title":"Doubly Mild Generalization for Offline Reinforcement Learning","date":"2024-11-12","arxiv_id":"2411.07934","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/doubly-mild-generalization-for-offline#ran","syntology_url":"https://syntology.ai/paper/2411.07934","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.07934"}},"official":{"repos":["maoyixiu/dmg"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tipo-text-to-image-with-text-presampling-for","slug":"tipo-text-to-image-with-text-presampling-for","title":"TIPO: Text to Image with Text Presampling for Prompt Optimization","date":"2024-11-12","arxiv_id":"2411.08127","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-quantum-tiq-taq","slug":"reinforcement-learning-for-quantum-tiq-taq","title":"Reinforcement learning for Quantum Tiq-Taq-Toe","date":"2024-11-10","arxiv_id":"2411.06429","repositories_listed":1,"syntology":null},{"url":"/paper/enabling-adaptive-agent-training-in-open","slug":"enabling-adaptive-agent-training-in-open","title":"Enabling Adaptive Agent Training in Open-Ended Simulators by Targeting Diversity","date":"2024-11-07","arxiv_id":"2411.04466","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/enabling-adaptive-agent-training-in-open#ran","syntology_url":"https://syntology.ai/paper/2411.04466","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.04466"}},"official":{"repos":["robbycostales/diva"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/noisy-zero-shot-coordination-breaking-the","slug":"noisy-zero-shot-coordination-breaking-the","title":"Noisy Zero-Shot Coordination: Breaking The Common Knowledge Assumption In Zero-Shot Coordination Games","date":"2024-11-07","arxiv_id":"2411.04976","repositories_listed":1,"syntology":null},{"url":"/paper/think-smart-act-smarl-analyzing-probabilistic","slug":"think-smart-act-smarl-analyzing-probabilistic","title":"Think Smart, Act SMARL! Analyzing Probabilistic Logic Shields for Multi-Agent Reinforcement Learning","date":"2024-11-07","arxiv_id":"2411.04867","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-transfer-reinforcement-learning","slug":"hybrid-transfer-reinforcement-learning","title":"Hybrid Transfer Reinforcement Learning: Provable Sample Efficiency from Shifted-Dynamics Data","date":"2024-11-06","arxiv_id":"2411.03810","repositories_listed":1,"syntology":null},{"url":"/paper/an-open-source-sim2real-approach-for-sensor","slug":"an-open-source-sim2real-approach-for-sensor","title":"An Open-source Sim2Real Approach for Sensor-independent Robot Navigation in a Grid","date":"2024-11-05","arxiv_id":"2411.03494","repositories_listed":1,"syntology":null},{"url":"/paper/risk-sensitive-control-as-inference-with","slug":"risk-sensitive-control-as-inference-with","title":"Risk-sensitive control as inference with Rényi divergence","date":"2024-11-04","arxiv_id":"2411.01827","repositories_listed":1,"syntology":null},{"url":"/paper/simulation-of-nanorobots-with-artificial","slug":"simulation-of-nanorobots-with-artificial","title":"Simulation of Nanorobots with Artificial Intelligence and Reinforcement Learning for Advanced Cancer Cell Detection and Tracking","date":"2024-11-04","arxiv_id":"2411.02345","repositories_listed":1,"syntology":null},{"url":"/paper/stepcountjitai-simulation-environment-for-rl","slug":"stepcountjitai-simulation-environment-for-rl","title":"StepCountJITAI: simulation environment for RL with application to physical activity adaptive intervention","date":"2024-11-01","arxiv_id":"2411.00336","repositories_listed":1,"syntology":null},{"url":"/paper/uncertainty-based-offline-variational","slug":"uncertainty-based-offline-variational","title":"Uncertainty-based Offline Variational Bayesian Reinforcement Learning for Robustness under Diverse Data Corruptions","date":"2024-11-01","arxiv_id":"2411.00465","repositories_listed":1,"syntology":{"n":13,"n_ran":4,"n_constructed":2,"n_ran_checked":2,"n_instrument":2,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/uncertainty-based-offline-variational#ran","syntology_url":"https://syntology.ai/paper/2411.00465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00465"}},"official":{"repos":["MIRALab-USTC/RL-TRACER"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/a-non-monolithic-policy-approach-of-offline","slug":"a-non-monolithic-policy-approach-of-offline","title":"A Non-Monolithic Policy Approach of Offline-to-Online Reinforcement Learning","date":"2024-10-31","arxiv_id":"2410.23737","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-gradients-as-vitamin","slug":"reinforcement-learning-gradients-as-vitamin","title":"Reinforcement Learning Gradients as Vitamin for Online Finetuning Decision Transformers","date":"2024-10-31","arxiv_id":"2410.24108","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reinforcement-learning-gradients-as-vitamin#ran","syntology_url":"https://syntology.ai/paper/2410.24108","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.24108"}},"official":{"repos":["kaiyan289/rl_as_vitamin_for_online_decision_transformers"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/teaching-embodied-reinforcement-learning","slug":"teaching-embodied-reinforcement-learning","title":"Teaching Embodied Reinforcement Learning Agents: Informativeness and Diversity of Language Use","date":"2024-10-31","arxiv_id":"2410.24218","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/teaching-embodied-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2410.24218","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.24218"}},"official":{"repos":["sled-group/teachable_rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/zonal-rl-rrt-integrated-rl-rrt-path-planning","slug":"zonal-rl-rrt-integrated-rl-rrt-path-planning","title":"Zonal RL-RRT: Integrated RL-RRT Path Planning with Collision Probability and Zone Connectivity","date":"2024-10-31","arxiv_id":"2410.24205","repositories_listed":1,"syntology":null},{"url":"/paper/kinetix-investigating-the-training-of-general","slug":"kinetix-investigating-the-training-of-general","title":"Kinetix: Investigating the Training of General Agents through Open-Ended Physics-Based Control Tasks","date":"2024-10-30","arxiv_id":"2410.23208","repositories_listed":1,"syntology":{"n":33,"n_ran":24,"n_constructed":6,"n_ran_checked":15,"n_instrument":9,"n_unverified":9,"n_honours":5,"n_violates":4,"n_no_contract":6,"n_pointer_only":0,"phrase":"24 ran (of which 6 constructed an object rather than computing a result; 15 with no instrument failure: 5 honoured, 4 violated, 6 with no contract checked; 9 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/kinetix-investigating-the-training-of-general#ran","syntology_url":"https://syntology.ai/paper/2410.23208","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.23208"}},"official":{"repos":["michaeltmatthews/jax2d"],"state":"official (archive's flag): 24 ran","n_ran":24,"n_constructed":6,"n_ran_no_instrument_failure":15,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-behavior-distillation","slug":"offline-behavior-distillation","title":"Offline Behavior Distillation","date":"2024-10-30","arxiv_id":"2410.22728","repositories_listed":1,"syntology":{"n":16,"n_ran":9,"n_constructed":1,"n_ran_checked":6,"n_instrument":3,"n_unverified":7,"n_honours":3,"n_violates":0,"n_no_contract":3,"n_pointer_only":16,"phrase":"9 ran (of which 1 constructed an object rather than computing a result; 6 with no instrument failure: 3 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/offline-behavior-distillation#ran","syntology_url":"https://syntology.ai/paper/2410.22728","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22728"}},"official":{"repos":["leaveslei/obd"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":1,"n_ran_no_instrument_failure":6,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/online-intrinsic-rewards-for-decision-making","slug":"online-intrinsic-rewards-for-decision-making","title":"Online Intrinsic Rewards for Decision Making Agents from Large Language Model Feedback","date":"2024-10-30","arxiv_id":"2410.23022","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":12,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/online-intrinsic-rewards-for-decision-making#ran","syntology_url":"https://syntology.ai/paper/2410.23022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.23022"}},"official":{"repos":["facebookresearch/oni"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/a-large-recurrent-action-model-xlstm-enables","slug":"a-large-recurrent-action-model-xlstm-enables","title":"A Large Recurrent Action Model: xLSTM enables Fast Inference for Robotics Tasks","date":"2024-10-29","arxiv_id":"2410.22391","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-large-recurrent-action-model-xlstm-enables#ran","syntology_url":"https://syntology.ai/paper/2410.22391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22391"}},"official":{"repos":["ml-jku/lram"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-successor-features-the-simple-way","slug":"learning-successor-features-the-simple-way","title":"Learning Successor Features the Simple Way","date":"2024-10-29","arxiv_id":"2410.22133","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-successor-features-the-simple-way#ran","syntology_url":"https://syntology.ai/paper/2410.22133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22133"}},"official":{"repos":["raymondchua/simple_successor_features"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pc-gym-benchmark-environments-for-process","slug":"pc-gym-benchmark-environments-for-process","title":"PC-Gym: Benchmark Environments For Process Control Problems","date":"2024-10-29","arxiv_id":"2410.22093","repositories_listed":1,"syntology":null},{"url":"/paper/fairstream-fair-multimedia-streaming","slug":"fairstream-fair-multimedia-streaming","title":"FairStream: Fair Multimedia Streaming Benchmark for Reinforcement Learning Agents","date":"2024-10-28","arxiv_id":"2410.21029","repositories_listed":1,"syntology":null},{"url":"/paper/longreward-improving-long-context-large","slug":"longreward-improving-long-context-large","title":"LongReward: Improving Long-context Large Language Models with AI Feedback","date":"2024-10-28","arxiv_id":"2410.21252","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longreward-improving-long-context-large#ran","syntology_url":"https://syntology.ai/paper/2410.21252","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21252"}},"official":{"repos":["THUDM/LongReward"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/odrl-a-benchmark-for-off-dynamics","slug":"odrl-a-benchmark-for-off-dynamics","title":"ODRL: A Benchmark for Off-Dynamics Reinforcement Learning","date":"2024-10-28","arxiv_id":"2410.20750","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/odrl-a-benchmark-for-off-dynamics#ran","syntology_url":"https://syntology.ai/paper/2410.20750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20750"}},"official":{"repos":["offdynamicsrl/off-dynamics-rl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ogbench-benchmarking-offline-goal-conditioned","slug":"ogbench-benchmarking-offline-goal-conditioned","title":"OGBench: Benchmarking Offline Goal-Conditioned RL","date":"2024-10-26","arxiv_id":"2410.20092","repositories_listed":1,"syntology":null},{"url":"/paper/agentforge-a-flexible-low-code-platform-for","slug":"agentforge-a-flexible-low-code-platform-for","title":"AgentForge: A Flexible Low-Code Platform for Reinforcement Learning Agent Design","date":"2024-10-25","arxiv_id":"2410.19528","repositories_listed":1,"syntology":null},{"url":"/paper/offline-reinforcement-learning-with-ood-state","slug":"offline-reinforcement-learning-with-ood-state","title":"Offline Reinforcement Learning with OOD State Correction and OOD Action Suppression","date":"2024-10-25","arxiv_id":"2410.19400","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/offline-reinforcement-learning-with-ood-state#ran","syntology_url":"https://syntology.ai/paper/2410.19400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.19400"}},"official":{"repos":["MAOYIXIU/SCAS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-versatile-skills-with-curriculum","slug":"learning-versatile-skills-with-curriculum","title":"Learning Versatile Skills with Curriculum Masking","date":"2024-10-23","arxiv_id":"2410.17744","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-versatile-skills-with-curriculum#ran","syntology_url":"https://syntology.ai/paper/2410.17744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17744"}},"official":{"repos":["yaotang23/currmask"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-skills-from-unlabeled-prior-data","slug":"leveraging-skills-from-unlabeled-prior-data","title":"Leveraging Skills from Unlabeled Prior Data for Efficient Online Exploration","date":"2024-10-23","arxiv_id":"2410.18076","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/leveraging-skills-from-unlabeled-prior-data#ran","syntology_url":"https://syntology.ai/paper/2410.18076","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18076"}},"official":{"repos":["rail-berkeley/supe"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-rl-based-llm-training-for-formal","slug":"exploring-rl-based-llm-training-for-formal","title":"Exploring RL-based LLM Training for Formal Language Tasks with Programmed Rewards","date":"2024-10-22","arxiv_id":"2410.17126","repositories_listed":1,"syntology":null},{"url":"/paper/integrating-reinforcement-learning-with","slug":"integrating-reinforcement-learning-with","title":"Integrating Reinforcement Learning with Foundation Models for Autonomous Robotics: Methods and Perspectives","date":"2024-10-21","arxiv_id":"2410.16411","repositories_listed":1,"syntology":null},{"url":"/paper/reinforced-imitative-trajectory-planning-for","slug":"reinforced-imitative-trajectory-planning-for","title":"Reinforced Imitative Trajectory Planning for Urban Automated Driving","date":"2024-10-21","arxiv_id":"2410.15607","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-dynamic-memory","slug":"reinforcement-learning-for-dynamic-memory","title":"Reinforcement Learning for Dynamic Memory Allocation","date":"2024-10-20","arxiv_id":"2410.15492","repositories_listed":1,"syntology":null},{"url":"/paper/intersectionzoo-eco-driving-for-benchmarking","slug":"intersectionzoo-eco-driving-for-benchmarking","title":"IntersectionZoo: Eco-driving for Benchmarking Multi-Agent Contextual Reinforcement Learning","date":"2024-10-19","arxiv_id":"2410.15221","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/intersectionzoo-eco-driving-for-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2410.15221","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.15221"}},"official":{"repos":["mit-wu-lab/IntersectionZoo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/streaming-deep-reinforcement-learning-finally","slug":"streaming-deep-reinforcement-learning-finally","title":"Streaming Deep Reinforcement Learning Finally Works","date":"2024-10-18","arxiv_id":"2410.14606","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/streaming-deep-reinforcement-learning-finally#ran","syntology_url":"https://syntology.ai/paper/2410.14606","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14606"}},"official":{"repos":["mohmdelsayed/streaming-drl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-tuning-discrete-diffusion-models-via","slug":"fine-tuning-discrete-diffusion-models-via","title":"Fine-Tuning Discrete Diffusion Models via Reward Optimization with Applications to DNA and Protein Design","date":"2024-10-17","arxiv_id":"2410.13643","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/fine-tuning-discrete-diffusion-models-via#ran","syntology_url":"https://syntology.ai/paper/2410.13643","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13643"}},"official":{"repos":["chenyuwang-monica/drakes"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/orso-accelerating-reward-design-via-online","slug":"orso-accelerating-reward-design-via-online","title":"ORSO: Accelerating Reward Design via Online Reward Selection and Policy Optimization","date":"2024-10-17","arxiv_id":"2410.13837","repositories_listed":1,"syntology":null},{"url":"/paper/sliding-puzzles-gym-a-scalable-benchmark-for","slug":"sliding-puzzles-gym-a-scalable-benchmark-for","title":"Sliding Puzzles Gym: A Scalable Benchmark for State Representation in Visual Reinforcement Learning","date":"2024-10-17","arxiv_id":"2410.14038","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sliding-puzzles-gym-a-scalable-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2410.14038","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14038"}},"official":{"repos":["bryanoliveira/sliding-puzzles-gym"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-euclidean-data","slug":"reinforcement-learning-with-euclidean-data","title":"Reinforcement Learning with Euclidean Data Augmentation for State-Based Continuous Control","date":"2024-10-16","arxiv_id":"2410.12983","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-reinforcement-learning-with-5","slug":"sample-efficient-reinforcement-learning-with-5","title":"Sample-Efficient Reinforcement Learning with Temporal Logic Objectives: Leveraging the Task Specification to Guide Exploration","date":"2024-10-16","arxiv_id":"2410.12136","repositories_listed":1,"syntology":null},{"url":"/paper/bayes-adaptive-monte-carlo-tree-search-for","slug":"bayes-adaptive-monte-carlo-tree-search-for","title":"Bayes Adaptive Monte Carlo Tree Search for Offline Model-based Reinforcement Learning","date":"2024-10-15","arxiv_id":"2410.11234","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bayes-adaptive-monte-carlo-tree-search-for#ran","syntology_url":"https://syntology.ai/paper/2410.11234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11234"}},"official":{"repos":["lucascjysdl/offline-rl-kit"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-dt-offline-meta-rl-as-conditional","slug":"meta-dt-offline-meta-rl-as-conditional","title":"Meta-DT: Offline Meta-RL as Conditional Sequence Modeling with World Model Disentanglement","date":"2024-10-15","arxiv_id":"2410.11448","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/meta-dt-offline-meta-rl-as-conditional#ran","syntology_url":"https://syntology.ai/paper/2410.11448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11448"}},"official":{"repos":["nju-rl/meta-dt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/safety-filtering-while-training-improving-the","slug":"safety-filtering-while-training-improving-the","title":"Safety Filtering While Training: Improving the Performance and Sample Efficiency of Reinforcement Learning Agents","date":"2024-10-15","arxiv_id":"2410.11671","repositories_listed":1,"syntology":null},{"url":"/paper/improving-generalization-on-the-procgen","slug":"improving-generalization-on-the-procgen","title":"Improving Generalization on the ProcGen Benchmark with Simple Architectural Changes and Scale","date":"2024-10-13","arxiv_id":"2410.10905","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-generalization-on-the-procgen#ran","syntology_url":"https://syntology.ai/paper/2410.10905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10905"}},"official":{"repos":["anndvision/vsop-3d"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sapient-mastering-multi-turn-conversational","slug":"sapient-mastering-multi-turn-conversational","title":"SAPIENT: Mastering Multi-turn Conversational Recommendation with Strategic Planning and Monte Carlo Tree Search","date":"2024-10-12","arxiv_id":"2410.09580","repositories_listed":1,"syntology":null},{"url":"/paper/drama-mamba-enabled-model-based-reinforcement","slug":"drama-mamba-enabled-model-based-reinforcement","title":"Drama: Mamba-Enabled Model-Based Reinforcement Learning Is Sample and Parameter Efficient","date":"2024-10-11","arxiv_id":"2410.08893","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/drama-mamba-enabled-model-based-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2410.08893","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08893"}},"official":{"repos":["realwenlongwang/drama"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/overcoming-slow-decision-frequencies-in","slug":"overcoming-slow-decision-frequencies-in","title":"Overcoming Slow Decision Frequencies in Continuous Control: Model-Based Sequence Reinforcement Learning for Model-Free Control","date":"2024-10-11","arxiv_id":"2410.08979","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":3,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/overcoming-slow-decision-frequencies-in#ran","syntology_url":"https://syntology.ai/paper/2410.08979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08979"}},"official":{"repos":["dee0512/Temporally-Layered-Architecture"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"url":"/paper/reinforcement-learning-for-control-of-non","slug":"reinforcement-learning-for-control-of-non","title":"Reinforcement Learning for Control of Non-Markovian Cellular Population Dynamics","date":"2024-10-11","arxiv_id":"2410.08439","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-for-control-of-non#ran","syntology_url":"https://syntology.ai/paper/2410.08439","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08439"}},"official":{"repos":["JacobHA/RL4Dosing"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-natural-language-based-strategies","slug":"exploring-natural-language-based-strategies","title":"Exploring Natural Language-Based Strategies for Efficient Number Learning in Children through Reinforcement Learning","date":"2024-10-10","arxiv_id":"2410.08334","repositories_listed":1,"syntology":null},{"url":"/paper/stableprompt-automatic-prompt-tuning-using","slug":"stableprompt-automatic-prompt-tuning-using","title":"StablePrompt: Automatic Prompt Tuning using Reinforcement Learning for Large Language Models","date":"2024-10-10","arxiv_id":"2410.07652","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/stableprompt-automatic-prompt-tuning-using#ran","syntology_url":"https://syntology.ai/paper/2410.07652","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07652"}},"official":{"repos":["kmc0207/Stableprompt"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/crafting-desirable-climate-trajectories-with","slug":"crafting-desirable-climate-trajectories-with","title":"Crafting desirable climate trajectories with RL explored socio-environmental simulations","date":"2024-10-09","arxiv_id":"2410.07287","repositories_listed":1,"syntology":null},{"url":"/paper/retrieval-augmented-decision-transformer","slug":"retrieval-augmented-decision-transformer","title":"Retrieval-Augmented Decision Transformer: External Memory for In-context RL","date":"2024-10-09","arxiv_id":"2410.07071","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/retrieval-augmented-decision-transformer#ran","syntology_url":"https://syntology.ai/paper/2410.07071","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07071"}},"official":{"repos":["ml-jku/RA-DT"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/coevolving-with-the-other-you-fine-tuning-llm","slug":"coevolving-with-the-other-you-fine-tuning-llm","title":"Coevolving with the Other You: Fine-Tuning LLM with Sequential Cooperative Multi-Agent Reinforcement Learning","date":"2024-10-08","arxiv_id":"2410.06101","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/coevolving-with-the-other-you-fine-tuning-llm#ran","syntology_url":"https://syntology.ai/paper/2410.06101","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.06101"}},"official":{"repos":["Harry67Hu/CORY"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/greenlight-gym-a-reinforcement-learning","slug":"greenlight-gym-a-reinforcement-learning","title":"GreenLight-Gym: Reinforcement learning benchmark environment for control of greenhouse production systems","date":"2024-10-06","arxiv_id":"2410.05336","repositories_listed":1,"syntology":null},{"url":"/paper/improved-off-policy-reinforcement-learning-in","slug":"improved-off-policy-reinforcement-learning-in","title":"Improved Off-policy Reinforcement Learning in Biological Sequence Design","date":"2024-10-06","arxiv_id":"2410.04461","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improved-off-policy-reinforcement-learning-in#ran","syntology_url":"https://syntology.ai/paper/2410.04461","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04461"}},"official":{"repos":["hyeonahkimm/delta_cs"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/improving-portfolio-optimization-results-with","slug":"improving-portfolio-optimization-results-with","title":"Improving Portfolio Optimization Results with Bandit Networks","date":"2024-10-05","arxiv_id":"2410.04217","repositories_listed":1,"syntology":null},{"url":"/paper/closd-closing-the-loop-between-simulation-and","slug":"closd-closing-the-loop-between-simulation-and","title":"CLoSD: Closing the Loop between Simulation and Diffusion for multi-task character control","date":"2024-10-04","arxiv_id":"2410.03441","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/closd-closing-the-loop-between-simulation-and#ran","syntology_url":"https://syntology.ai/paper/2410.03441","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03441"}},"official":{"repos":["GuyTevet/CLoSD"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/mitigating-adversarial-perturbations-for-deep","slug":"mitigating-adversarial-perturbations-for-deep","title":"Mitigating Adversarial Perturbations for Deep Reinforcement Learning via Vector Quantization","date":"2024-10-04","arxiv_id":"2410.03376","repositories_listed":1,"syntology":null}],"record_sha256":"c353ef8167e1deec453c7780851477a6cc6ed52dea3d692c92a31df41407ef7c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}