{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/11","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":11,"pages_in_order":135,"rows_per_page":100,"rows":[1001,1100],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/10","next":"/task/reinforcement-learning-2/papers/12","papers":[{"url":"/paper/inverse-reinforcement-learning-via-convex","slug":"inverse-reinforcement-learning-via-convex","title":"Inverse Reinforcement Learning via Convex Optimization","date":"2025-01-27","arxiv_id":"2501.15957","repositories_listed":1,"syntology":null},{"url":"/paper/multi-objective-reinforcement-learning-for-2","slug":"multi-objective-reinforcement-learning-for-2","title":"Multi-Objective Reinforcement Learning for Power Grid Topology Control","date":"2025-01-27","arxiv_id":"2502.00040","repositories_listed":1,"syntology":null},{"url":"/paper/refill-reinforcement-learning-for-fill-in","slug":"refill-reinforcement-learning-for-fill-in","title":"ReFill: Reinforcement Learning for Fill-In Minimization","date":"2025-01-27","arxiv_id":"2501.16130","repositories_listed":1,"syntology":null},{"url":"/paper/upside-down-reinforcement-learning-with","slug":"upside-down-reinforcement-learning-with","title":"Upside Down Reinforcement Learning with Policy Generators","date":"2025-01-27","arxiv_id":"2501.16288","repositories_listed":1,"syntology":null},{"url":"/paper/expert-free-online-transfer-learning-in-multi-1","slug":"expert-free-online-transfer-learning-in-multi-1","title":"Expert-Free Online Transfer Learning in Multi-Agent Reinforcement Learning","date":"2025-01-26","arxiv_id":"2501.15495","repositories_listed":1,"syntology":null},{"url":"/paper/divergence-augmented-policy-optimization-1","slug":"divergence-augmented-policy-optimization-1","title":"Divergence-Augmented Policy Optimization","date":"2025-01-25","arxiv_id":"2501.15034","repositories_listed":1,"syntology":null},{"url":"/paper/improving-retrieval-augmented-generation","slug":"improving-retrieval-augmented-generation","title":"Improving Retrieval-Augmented Generation through Multi-Agent Reinforcement Learning","date":"2025-01-25","arxiv_id":"2501.15228","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-retrieval-augmented-generation#ran","syntology_url":"https://syntology.ai/paper/2501.15228","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.15228"}},"official":{"repos":["chenyiqun/mmoa-rag"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/agentrec-agent-recommendation-using-sentence","slug":"agentrec-agent-recommendation-using-sentence","title":"AgentRec: Agent Recommendation Using Sentence Embeddings Aligned to Human Feedback","date":"2025-01-23","arxiv_id":"2501.13333","repositories_listed":1,"syntology":null},{"url":"/paper/scalable-safe-multi-agent-reinforcement","slug":"scalable-safe-multi-agent-reinforcement","title":"Scalable Safe Multi-Agent Reinforcement Learning for Multi-Agent System","date":"2025-01-23","arxiv_id":"2501.13727","repositories_listed":1,"syntology":null},{"url":"/paper/utilizing-evolution-strategies-to-train","slug":"utilizing-evolution-strategies-to-train","title":"Utilizing Evolution Strategies to Train Transformers in Reinforcement Learning","date":"2025-01-23","arxiv_id":"2501.13883","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/utilizing-evolution-strategies-to-train#ran","syntology_url":"https://syntology.ai/paper/2501.13883","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.13883"}},"official":{"repos":["mafi412/evolution-strategies-and-decision-transformers"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/wfcrl-a-multi-agent-reinforcement-learning","slug":"wfcrl-a-multi-agent-reinforcement-learning","title":"WFCRL: A Multi-Agent Reinforcement Learning Benchmark for Wind Farm Control","date":"2025-01-23","arxiv_id":"2501.13592","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-data-exploitation-in-deep","slug":"adaptive-data-exploitation-in-deep","title":"Adaptive Data Exploitation in Deep Reinforcement Learning","date":"2025-01-22","arxiv_id":"2501.12620","repositories_listed":1,"syntology":null},{"url":"/paper/srmt-shared-memory-for-multi-agent-lifelong","slug":"srmt-shared-memory-for-multi-agent-lifelong","title":"SRMT: Shared Memory for Multi-agent Lifelong Pathfinding","date":"2025-01-22","arxiv_id":"2501.13200","repositories_listed":1,"syntology":null},{"url":"/paper/tackling-uncertainties-in-multi-agent","slug":"tackling-uncertainties-in-multi-agent","title":"Tackling Uncertainties in Multi-Agent Reinforcement Learning through Integration of Agent Termination Dynamics","date":"2025-01-21","arxiv_id":"2501.12061","repositories_listed":1,"syntology":null},{"url":"/paper/curiosity-driven-reinforcement-learning-from","slug":"curiosity-driven-reinforcement-learning-from","title":"Curiosity-Driven Reinforcement Learning from Human Feedback","date":"2025-01-20","arxiv_id":"2501.11463","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/curiosity-driven-reinforcement-learning-from#ran","syntology_url":"https://syntology.ai/paper/2501.11463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.11463"}},"official":{"repos":["ernie-research/cd-rlhf"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pearl-preconditioner-enhancement-through","slug":"pearl-preconditioner-enhancement-through","title":"PEARL: Preconditioner Enhancement through Actor-critic Reinforcement Learning","date":"2025-01-18","arxiv_id":"2501.10750","repositories_listed":1,"syntology":null},{"url":"/paper/ansr-dt-an-adaptive-neuro-symbolic-learning","slug":"ansr-dt-an-adaptive-neuro-symbolic-learning","title":"ANSR-DT: An Adaptive Neuro-Symbolic Learning and Reasoning Framework for Digital Twins","date":"2025-01-15","arxiv_id":"2501.08561","repositories_listed":1,"syntology":null},{"url":"/paper/cuasmrl-optimizing-gpu-sass-schedules-via","slug":"cuasmrl-optimizing-gpu-sass-schedules-via","title":"CuAsmRL: Optimizing GPU SASS Schedules via Deep Reinforcement Learning","date":"2025-01-14","arxiv_id":"2501.08071","repositories_listed":1,"syntology":null},{"url":"/paper/a-hybrid-framework-for-reinsurance","slug":"a-hybrid-framework-for-reinsurance","title":"A Hybrid Framework for Reinsurance Optimization: Integrating Generative Models and Reinforcement Learning","date":"2025-01-11","arxiv_id":"2501.06404","repositories_listed":1,"syntology":null},{"url":"/paper/the-meta-representation-hypothesis","slug":"the-meta-representation-hypothesis","title":"Representation Convergence: Mutual Distillation is Secretly a Form of Regularization","date":"2025-01-05","arxiv_id":"2501.02481","repositories_listed":1,"syntology":null},{"url":"/paper/noise-resilient-symbolic-regression-with","slug":"noise-resilient-symbolic-regression-with","title":"Noise-Resilient Symbolic Regression with Dynamic Gating Reinforcement Learning","date":"2025-01-02","arxiv_id":"2501.01085","repositories_listed":1,"syntology":null},{"url":"/paper/hybridising-reinforcement-learning-and","slug":"hybridising-reinforcement-learning-and","title":"Hybridising Reinforcement Learning and Heuristics for Hierarchical Directed Arc Routing Problems","date":"2025-01-01","arxiv_id":"2501.00852","repositories_listed":1,"syntology":null},{"url":"/paper/lease-offline-preference-based-reinforcement","slug":"lease-offline-preference-based-reinforcement","title":"LEASE: Offline Preference-based Reinforcement Learning with High Sample Efficiency","date":"2024-12-30","arxiv_id":"2412.21001","repositories_listed":1,"syntology":null},{"url":"/paper/diminishing-return-of-value-expansion-methods-1","slug":"diminishing-return-of-value-expansion-methods-1","title":"Diminishing Return of Value Expansion Methods","date":"2024-12-29","arxiv_id":"2412.20537","repositories_listed":1,"syntology":null},{"url":"/paper/numerical-solutions-of-fixed-points-in-two","slug":"numerical-solutions-of-fixed-points-in-two","title":"Numerical solutions of fixed points in two-dimensional Kuramoto-Sivashinsky equation expedited by reinforcement learning","date":"2024-12-27","arxiv_id":"2501.00046","repositories_listed":1,"syntology":null},{"url":"/paper/constraint-adaptive-policy-switching-for","slug":"constraint-adaptive-policy-switching-for","title":"Constraint-Adaptive Policy Switching for Offline Safe Reinforcement Learning","date":"2024-12-25","arxiv_id":"2412.18946","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/constraint-adaptive-policy-switching-for#ran","syntology_url":"https://syntology.ai/paper/2412.18946","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.18946"}},"official":{"repos":["yassinech/caps"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-optimization-of-portfolio-allocation","slug":"dynamic-optimization-of-portfolio-allocation","title":"A Deep Reinforcement Learning Framework for Dynamic Portfolio Optimization: Evidence from China's Stock Market","date":"2024-12-24","arxiv_id":"2412.18563","repositories_listed":1,"syntology":null},{"url":"/paper/llm-powered-user-simulator-for-recommender","slug":"llm-powered-user-simulator-for-recommender","title":"LLM-Powered User Simulator for Recommender System","date":"2024-12-22","arxiv_id":"2412.16984","repositories_listed":1,"syntology":null},{"url":"/paper/autonomous-option-invention-for-continual","slug":"autonomous-option-invention-for-continual","title":"Autonomous Option Invention for Continual Hierarchical Reinforcement Learning and Planning","date":"2024-12-20","arxiv_id":"2412.16395","repositories_listed":1,"syntology":null},{"url":"/paper/decoding-fairness-a-reinforcement-learning","slug":"decoding-fairness-a-reinforcement-learning","title":"Decoding fairness: a reinforcement learning perspective","date":"2024-12-20","arxiv_id":"2412.16249","repositories_listed":1,"syntology":null},{"url":"/paper/fedrlhf-a-convergence-guaranteed-federated","slug":"fedrlhf-a-convergence-guaranteed-federated","title":"FedRLHF: A Convergence-Guaranteed Federated Framework for Privacy-Preserving and Personalized RLHF","date":"2024-12-20","arxiv_id":"2412.15538","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-for-24","slug":"multi-agent-reinforcement-learning-for-24","title":"Multi Agent Reinforcement Learning for Sequential Satellite Assignment Problems","date":"2024-12-20","arxiv_id":"2412.15573","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-agent-reinforcement-learning-for-24#ran","syntology_url":"https://syntology.ai/paper/2412.15573","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.15573"}},"official":{"repos":["Rainlabuw/rl-enabled-distributed-assignment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-with-time-scale","slug":"deep-reinforcement-learning-with-time-scale","title":"Deep reinforcement learning with time-scale invariant memory","date":"2024-12-19","arxiv_id":"2412.15292","repositories_listed":1,"syntology":null},{"url":"/paper/offline-safe-reinforcement-learning-using","slug":"offline-safe-reinforcement-learning-using","title":"Offline Safe Reinforcement Learning Using Trajectory Classification","date":"2024-12-19","arxiv_id":"2412.15429","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/offline-safe-reinforcement-learning-using#ran","syntology_url":"https://syntology.ai/paper/2412.15429","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.15429"}},"official":{"repos":["zgong11/TraC"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/guiding-generative-protein-language-models","slug":"guiding-generative-protein-language-models","title":"Guiding Generative Protein Language Models with Reinforcement Learning","date":"2024-12-17","arxiv_id":"2412.12979","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/guiding-generative-protein-language-models#ran","syntology_url":"https://syntology.ai/paper/2412.12979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.12979"}},"official":{"repos":["ai4pdlab/dpo_plm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tilted-quantile-gradient-updates-for-quantile","slug":"tilted-quantile-gradient-updates-for-quantile","title":"Tilted Quantile Gradient Updates for Quantile-Constrained Reinforcement Learning","date":"2024-12-17","arxiv_id":"2412.13184","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-reward-design-for-reinforcement","slug":"adaptive-reward-design-for-reinforcement","title":"Adaptive Reward Design for Reinforcement Learning","date":"2024-12-14","arxiv_id":"2412.10917","repositories_listed":1,"syntology":null},{"url":"/paper/latent-safety-constrained-policy-approach-for","slug":"latent-safety-constrained-policy-approach-for","title":"Latent Safety-Constrained Policy Approach for Safe Offline Reinforcement Learning","date":"2024-12-11","arxiv_id":"2412.08794","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/latent-safety-constrained-policy-approach-for#ran","syntology_url":"https://syntology.ai/paper/2412.08794","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.08794"}},"official":{"repos":["PrajwalKoirala/LSPC-Safe-Offline-RL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-method-for-evaluating-hyperparameter","slug":"a-method-for-evaluating-hyperparameter","title":"A Method for Evaluating Hyperparameter Sensitivity in Reinforcement Learning","date":"2024-12-10","arxiv_id":"2412.07165","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-method-for-evaluating-hyperparameter#ran","syntology_url":"https://syntology.ai/paper/2412.07165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.07165"}},"official":{"repos":["jadkins99/hyperparameter_sensitivity"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/configx-modular-configuration-for","slug":"configx-modular-configuration-for","title":"ConfigX: Modular Configuration for Evolutionary Algorithms via Multitask Reinforcement Learning","date":"2024-12-10","arxiv_id":"2412.07507","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-policy-as-macro","slug":"reinforcement-learning-policy-as-macro","title":"Reinforcement Learning Policy as Macro Regulator Rather than Macro Placer","date":"2024-12-10","arxiv_id":"2412.07167","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reinforcement-learning-policy-as-macro#ran","syntology_url":"https://syntology.ai/paper/2412.07167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.07167"}},"official":{"repos":["lamda-bbo/macro-regulator"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/off-policy-maximum-entropy-rl-with-future","slug":"off-policy-maximum-entropy-rl-with-future","title":"Off-Policy Maximum Entropy RL with Future State and Action Visitation Measures","date":"2024-12-09","arxiv_id":"2412.06655","repositories_listed":1,"syntology":null},{"url":"/paper/simudice-offline-policy-optimization-through","slug":"simudice-offline-policy-optimization-through","title":"SimuDICE: Offline Policy Optimization Through World Model Updates and DICE Estimation","date":"2024-12-09","arxiv_id":"2412.06486","repositories_listed":1,"syntology":null},{"url":"/paper/classifier-free-guidance-in-llms-safety","slug":"classifier-free-guidance-in-llms-safety","title":"Classifier-free guidance in LLMs Safety","date":"2024-12-08","arxiv_id":"2412.06846","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/classifier-free-guidance-in-llms-safety#ran","syntology_url":"https://syntology.ai/paper/2412.06846","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06846"}},"official":{"repos":["rgsmirnov/cfg_safety_llm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-an-overview","slug":"reinforcement-learning-an-overview","title":"Reinforcement Learning: An Overview","date":"2024-12-06","arxiv_id":"2412.05265","repositories_listed":1,"syntology":null},{"url":"/paper/gram-generalization-in-deep-rl-with-a-robust","slug":"gram-generalization-in-deep-rl-with-a-robust","title":"GRAM: Generalization in Deep RL with a Robust Adaptation Module","date":"2024-12-05","arxiv_id":"2412.04323","repositories_listed":1,"syntology":null},{"url":"/paper/marvel-accelerating-safe-online-reinforcement","slug":"marvel-accelerating-safe-online-reinforcement","title":"Marvel: Accelerating Safe Online Reinforcement Learning with Finetuned Offline Policy","date":"2024-12-05","arxiv_id":"2412.04426","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-enhanced-llms-a-survey","slug":"reinforcement-learning-enhanced-llms-a-survey","title":"Reinforcement Learning Enhanced LLMs: A Survey","date":"2024-12-05","arxiv_id":"2412.10400","repositories_listed":1,"syntology":null},{"url":"/paper/conformal-symplectic-optimization-for-stable","slug":"conformal-symplectic-optimization-for-stable","title":"Conformal Symplectic Optimization for Stable Reinforcement Learning","date":"2024-12-03","arxiv_id":"2412.02291","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-plastic-waste-collection-in-water","slug":"optimizing-plastic-waste-collection-in-water","title":"Optimizing Plastic Waste Collection in Water Bodies Using Heterogeneous Autonomous Surface Vehicles with Deep Reinforcement Learning","date":"2024-12-03","arxiv_id":"2412.02316","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-generative-policies-a-simpler","slug":"revisiting-generative-policies-a-simpler","title":"Revisiting Generative Policies: A Simpler Reinforcement Learning Algorithmic Perspective","date":"2024-12-02","arxiv_id":"2412.01245","repositories_listed":1,"syntology":null},{"url":"/paper/towards-fault-tolerance-in-multi-agent","slug":"towards-fault-tolerance-in-multi-agent","title":"Towards Fault Tolerance in Multi-Agent Reinforcement Learning","date":"2024-11-30","arxiv_id":"2412.00534","repositories_listed":1,"syntology":null},{"url":"/paper/carel-instruction-guided-reinforcement","slug":"carel-instruction-guided-reinforcement","title":"CAREL: Instruction-guided reinforcement learning with cross-modal auxiliary objectives","date":"2024-11-29","arxiv_id":"2411.19787","repositories_listed":1,"syntology":null},{"url":"/paper/continual-deep-reinforcement-learning-with","slug":"continual-deep-reinforcement-learning-with","title":"Continual Deep Reinforcement Learning with Task-Agnostic Policy Distillation","date":"2024-11-25","arxiv_id":"2411.16532","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/continual-deep-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2411.16532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16532"}},"official":{"repos":["wabbajack1/tapd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-environments-for-vehicle-routing","slug":"multi-agent-environments-for-vehicle-routing","title":"Multi-Agent Environments for Vehicle Routing Problems","date":"2024-11-21","arxiv_id":"2411.14411","repositories_listed":1,"syntology":null},{"url":"/paper/natural-language-reinforcement-learning-1","slug":"natural-language-reinforcement-learning-1","title":"Natural Language Reinforcement Learning","date":"2024-11-21","arxiv_id":"2411.14251","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/natural-language-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2411.14251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.14251"}},"official":{"repos":["waterhorse1/natural-language-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/syllabus-portable-curricula-for-reinforcement","slug":"syllabus-portable-curricula-for-reinforcement","title":"Syllabus: Portable Curricula for Reinforcement Learning Agents","date":"2024-11-18","arxiv_id":"2411.11318","repositories_listed":1,"syntology":null},{"url":"/paper/an-investigation-of-offline-reinforcement","slug":"an-investigation-of-offline-reinforcement","title":"An Investigation of Offline Reinforcement Learning in Factorisable Action Spaces","date":"2024-11-17","arxiv_id":"2411.11088","repositories_listed":1,"syntology":null},{"url":"/paper/precision-focused-reinforcement-learning","slug":"precision-focused-reinforcement-learning","title":"Precision-Focused Reinforcement Learning Model for Robotic Object Pushing","date":"2024-11-13","arxiv_id":"2411.08622","repositories_listed":1,"syntology":null},{"url":"/paper/recommender-systems-and-reinforcement","slug":"recommender-systems-and-reinforcement","title":"Recommender systems and reinforcement learning for human-building interaction and context-aware support: A text mining-driven review of scientific literature","date":"2024-11-13","arxiv_id":"2411.08734","repositories_listed":1,"syntology":null},{"url":"/paper/doubly-mild-generalization-for-offline","slug":"doubly-mild-generalization-for-offline","title":"Doubly Mild Generalization for Offline Reinforcement Learning","date":"2024-11-12","arxiv_id":"2411.07934","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/doubly-mild-generalization-for-offline#ran","syntology_url":"https://syntology.ai/paper/2411.07934","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.07934"}},"official":{"repos":["maoyixiu/dmg"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/non-adversarial-inverse-reinforcement","slug":"non-adversarial-inverse-reinforcement","title":"Non-Adversarial Inverse Reinforcement Learning via Successor Feature Matching","date":"2024-11-11","arxiv_id":"2411.07007","repositories_listed":1,"syntology":{"n":24,"n_ran":19,"n_constructed":14,"n_ran_checked":15,"n_instrument":4,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":14,"n_pointer_only":0,"phrase":"19 ran (of which 14 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 1 violated, 14 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/non-adversarial-inverse-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2411.07007","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.07007"}},"official":{"repos":["arnavkj1995/sfm"],"state":"official (archive's flag): 19 ran","n_ran":19,"n_constructed":14,"n_ran_no_instrument_failure":15,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-quantum-tiq-taq","slug":"reinforcement-learning-for-quantum-tiq-taq","title":"Reinforcement learning for Quantum Tiq-Taq-Toe","date":"2024-11-10","arxiv_id":"2411.06429","repositories_listed":1,"syntology":null},{"url":"/paper/state-chrono-representation-for-enhancing","slug":"state-chrono-representation-for-enhancing","title":"State Chrono Representation for Enhancing Generalization in Reinforcement Learning","date":"2024-11-09","arxiv_id":"2411.06174","repositories_listed":1,"syntology":null},{"url":"/paper/tangled-program-graphs-as-an-alternative-to","slug":"tangled-program-graphs-as-an-alternative-to","title":"Tangled Program Graphs as an alternative to DRL-based control algorithms for UAVs","date":"2024-11-08","arxiv_id":"2411.05586","repositories_listed":1,"syntology":null},{"url":"/paper/constrained-latent-action-policies-for-model","slug":"constrained-latent-action-policies-for-model","title":"Constrained Latent Action Policies for Model-Based Offline Reinforcement Learning","date":"2024-11-07","arxiv_id":"2411.04562","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/constrained-latent-action-policies-for-model#ran","syntology_url":"https://syntology.ai/paper/2411.04562","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.04562"}},"official":{"repos":["marvinalles/c-lap"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/hypercube-policy-regularization-framework-for","slug":"hypercube-policy-regularization-framework-for","title":"Hypercube Policy Regularization Framework for Offline Reinforcement Learning","date":"2024-11-07","arxiv_id":"2411.04534","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-aware-resource-management-for-c-v2x","slug":"semantic-aware-resource-management-for-c-v2x","title":"Semantic-Aware Resource Management for C-V2X Platooning via Multi-Agent Reinforcement Learning","date":"2024-11-07","arxiv_id":"2411.04672","repositories_listed":1,"syntology":null},{"url":"/paper/think-smart-act-smarl-analyzing-probabilistic","slug":"think-smart-act-smarl-analyzing-probabilistic","title":"Think Smart, Act SMARL! Analyzing Probabilistic Logic Shields for Multi-Agent Reinforcement Learning","date":"2024-11-07","arxiv_id":"2411.04867","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-assist-humans-without-inferring","slug":"learning-to-assist-humans-without-inferring","title":"Learning to Assist Humans without Inferring Rewards","date":"2024-11-04","arxiv_id":"2411.02623","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/learning-to-assist-humans-without-inferring#ran","syntology_url":"https://syntology.ai/paper/2411.02623","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02623"}},"official":{"repos":["vivekmyers/empowerment_successor_representations"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/enhancing-chess-reinforcement-learning-with","slug":"enhancing-chess-reinforcement-learning-with","title":"Enhancing Chess Reinforcement Learning with Graph Representation","date":"2024-10-31","arxiv_id":"2410.23753","repositories_listed":1,"syntology":null},{"url":"/paper/from-easy-to-hard-tackling-quantum-problems","slug":"from-easy-to-hard-tackling-quantum-problems","title":"Reinforcement learning with learned gadgets to tackle hard quantum problems on real hardware","date":"2024-10-31","arxiv_id":"2411.00230","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/from-easy-to-hard-tackling-quantum-problems#ran","syntology_url":"https://syntology.ai/paper/2411.00230","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00230"}},"official":{"repos":["aqasch/gadget_rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/prosody-as-a-teaching-signal-for-agent","slug":"prosody-as-a-teaching-signal-for-agent","title":"Prosody as a Teaching Signal for Agent Learning: Exploratory Studies and Algorithmic Implications","date":"2024-10-31","arxiv_id":"2410.23554","repositories_listed":1,"syntology":null},{"url":"/paper/ra-pbrl-provably-efficient-risk-aware","slug":"ra-pbrl-provably-efficient-risk-aware","title":"RA-PbRL: Provably Efficient Risk-Aware Preference-Based Reinforcement Learning","date":"2024-10-31","arxiv_id":"2410.23569","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-gradients-as-vitamin","slug":"reinforcement-learning-gradients-as-vitamin","title":"Reinforcement Learning Gradients as Vitamin for Online Finetuning Decision Transformers","date":"2024-10-31","arxiv_id":"2410.24108","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reinforcement-learning-gradients-as-vitamin#ran","syntology_url":"https://syntology.ai/paper/2410.24108","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.24108"}},"official":{"repos":["kaiyan289/rl_as_vitamin_for_online_decision_transformers"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/econojax-a-fast-scalable-economic-simulation","slug":"econojax-a-fast-scalable-economic-simulation","title":"EconoJax: A Fast & Scalable Economic Simulation in Jax","date":"2024-10-29","arxiv_id":"2410.22165","repositories_listed":1,"syntology":null},{"url":"/paper/predicting-future-actions-of-reinforcement","slug":"predicting-future-actions-of-reinforcement","title":"Predicting Future Actions of Reinforcement Learning Agents","date":"2024-10-29","arxiv_id":"2410.22459","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/predicting-future-actions-of-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2410.22459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22459"}},"official":{"repos":["stephen-chung-mh/predict_action"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fairstream-fair-multimedia-streaming","slug":"fairstream-fair-multimedia-streaming","title":"FairStream: Fair Multimedia Streaming Benchmark for Reinforcement Learning Agents","date":"2024-10-28","arxiv_id":"2410.21029","repositories_listed":1,"syntology":null},{"url":"/paper/falcon-feedback-driven-adaptive-long-short","slug":"falcon-feedback-driven-adaptive-long-short","title":"FALCON: Feedback-driven Adaptive Long/short-term memory reinforced Coding Optimization system","date":"2024-10-28","arxiv_id":"2410.21349","repositories_listed":1,"syntology":null},{"url":"/paper/odrl-a-benchmark-for-off-dynamics","slug":"odrl-a-benchmark-for-off-dynamics","title":"ODRL: A Benchmark for Off-Dynamics Reinforcement Learning","date":"2024-10-28","arxiv_id":"2410.20750","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/odrl-a-benchmark-for-off-dynamics#ran","syntology_url":"https://syntology.ai/paper/2410.20750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20750"}},"official":{"repos":["offdynamicsrl/off-dynamics-rl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-modeling-with-weak-supervision-for","slug":"reward-modeling-with-weak-supervision-for","title":"Reward Modeling with Weak Supervision for Language Models","date":"2024-10-28","arxiv_id":"2410.20869","repositories_listed":1,"syntology":null},{"url":"/paper/robustness-and-generalization-in-quantum","slug":"robustness-and-generalization-in-quantum","title":"Robustness and Generalization in Quantum Reinforcement Learning via Lipschitz Regularization","date":"2024-10-28","arxiv_id":"2410.21117","repositories_listed":1,"syntology":null},{"url":"/paper/ogbench-benchmarking-offline-goal-conditioned","slug":"ogbench-benchmarking-offline-goal-conditioned","title":"OGBench: Benchmarking Offline Goal-Conditioned RL","date":"2024-10-26","arxiv_id":"2410.20092","repositories_listed":1,"syntology":null},{"url":"/paper/toward-finding-strong-pareto-optimal-policies","slug":"toward-finding-strong-pareto-optimal-policies","title":"Toward Finding Strong Pareto Optimal Policies in Multi-Agent Reinforcement Learning","date":"2024-10-25","arxiv_id":"2410.19372","repositories_listed":1,"syntology":null},{"url":"/paper/entity-based-reinforcement-learning-for","slug":"entity-based-reinforcement-learning-for","title":"Entity-based Reinforcement Learning for Autonomous Cyber Defence","date":"2024-10-23","arxiv_id":"2410.17647","repositories_listed":1,"syntology":null},{"url":"/paper/scalable-offline-reinforcement-learning-for","slug":"scalable-offline-reinforcement-learning-for","title":"Scalable Offline Reinforcement Learning for Mean Field Games","date":"2024-10-23","arxiv_id":"2410.17898","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-two-player-performance-through","slug":"enhancing-two-player-performance-through","title":"Enhancing Two-Player Performance Through Single-Player Knowledge Transfer: An Empirical Study on Atari 2600 Games","date":"2024-10-22","arxiv_id":"2410.16653","repositories_listed":1,"syntology":null},{"url":"/paper/evolution-with-opponent-learning-awareness","slug":"evolution-with-opponent-learning-awareness","title":"Evolution of Societies via Reinforcement Learning","date":"2024-10-22","arxiv_id":"2410.17466","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-rl-based-llm-training-for-formal","slug":"exploring-rl-based-llm-training-for-formal","title":"Exploring RL-based LLM Training for Formal Language Tasks with Programmed Rewards","date":"2024-10-22","arxiv_id":"2410.17126","repositories_listed":1,"syntology":null},{"url":"/paper/navigating-noisy-feedback-enhancing","slug":"navigating-noisy-feedback-enhancing","title":"Navigating Noisy Feedback: Enhancing Reinforcement Learning with Error-Prone Language Models","date":"2024-10-22","arxiv_id":"2410.17389","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/navigating-noisy-feedback-enhancing#ran","syntology_url":"https://syntology.ai/paper/2410.17389","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17389"}},"official":{"repos":["sy-shi/RLAIF_ScoreDiff"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforced-imitative-trajectory-planning-for","slug":"reinforced-imitative-trajectory-planning-for","title":"Reinforced Imitative Trajectory Planning for Urban Automated Driving","date":"2024-10-21","arxiv_id":"2410.15607","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-dynamic-memory","slug":"reinforcement-learning-for-dynamic-memory","title":"Reinforcement Learning for Dynamic Memory Allocation","date":"2024-10-20","arxiv_id":"2410.15492","repositories_listed":1,"syntology":null},{"url":"/paper/cooperation-and-fairness-in-multi-agent","slug":"cooperation-and-fairness-in-multi-agent","title":"Cooperation and Fairness in Multi-Agent Reinforcement Learning","date":"2024-10-19","arxiv_id":"2410.14916","repositories_listed":1,"syntology":null},{"url":"/paper/intersectionzoo-eco-driving-for-benchmarking","slug":"intersectionzoo-eco-driving-for-benchmarking","title":"IntersectionZoo: Eco-driving for Benchmarking Multi-Agent Contextual Reinforcement Learning","date":"2024-10-19","arxiv_id":"2410.15221","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/intersectionzoo-eco-driving-for-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2410.15221","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.15221"}},"official":{"repos":["mit-wu-lab/IntersectionZoo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-deep-reinforcement-learning-for-1","slug":"benchmarking-deep-reinforcement-learning-for-1","title":"Benchmarking Deep Reinforcement Learning for Navigation in Denied Sensor Environments","date":"2024-10-18","arxiv_id":"2410.14616","repositories_listed":1,"syntology":null},{"url":"/paper/streaming-deep-reinforcement-learning-finally","slug":"streaming-deep-reinforcement-learning-finally","title":"Streaming Deep Reinforcement Learning Finally Works","date":"2024-10-18","arxiv_id":"2410.14606","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/streaming-deep-reinforcement-learning-finally#ran","syntology_url":"https://syntology.ai/paper/2410.14606","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14606"}},"official":{"repos":["mohmdelsayed/streaming-drl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-online-5","slug":"deep-reinforcement-learning-for-online-5","title":"Deep Reinforcement Learning for Online Optimal Execution Strategies","date":"2024-10-17","arxiv_id":"2410.13493","repositories_listed":1,"syntology":null},{"url":"/paper/bayes-adaptive-monte-carlo-tree-search-for","slug":"bayes-adaptive-monte-carlo-tree-search-for","title":"Bayes Adaptive Monte Carlo Tree Search for Offline Model-based Reinforcement Learning","date":"2024-10-15","arxiv_id":"2410.11234","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bayes-adaptive-monte-carlo-tree-search-for#ran","syntology_url":"https://syntology.ai/paper/2410.11234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11234"}},"official":{"repos":["lucascjysdl/offline-rl-kit"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-shot-model-based-reinforcement-learning","slug":"zero-shot-model-based-reinforcement-learning","title":"Zero-shot Model-based Reinforcement Learning using Large Language Models","date":"2024-10-15","arxiv_id":"2410.11711","repositories_listed":1,"syntology":null},{"url":"/paper/stable-hadamard-memory-revitalizing-memory","slug":"stable-hadamard-memory-revitalizing-memory","title":"Stable Hadamard Memory: Revitalizing Memory-Augmented Agents for Reinforcement Learning","date":"2024-10-14","arxiv_id":"2410.10132","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stable-hadamard-memory-revitalizing-memory#ran","syntology_url":"https://syntology.ai/paper/2410.10132","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10132"}},"official":null}}],"record_sha256":"4877c79316db944492269a3896728c77fbf5cc17ba101f53e40d0a97b7cd36b1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}