{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/12","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":12,"pages_in_order":132,"rows_per_page":100,"rows":[1101,1200],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/11","next":"/task/reinforcement-learning/papers/13","papers":[{"url":"/paper/neural-guided-equation-discovery","slug":"neural-guided-equation-discovery","title":"Neural-Guided Equation Discovery","date":"2025-03-21","arxiv_id":"2503.16953","repositories_listed":1,"syntology":null},{"url":"/paper/cls-rl-image-classification-with-rule-based","slug":"cls-rl-image-classification-with-rule-based","title":"Think or Not Think: A Study of Explicit Thinking in Rule-Based Visual Reinforcement Fine-Tuning","date":"2025-03-20","arxiv_id":"2503.16188","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cls-rl-image-classification-with-rule-based#ran","syntology_url":"https://syntology.ai/paper/2503.16188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16188"}},"official":{"repos":["minglllli/CLS-RL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-based-heuristics-to","slug":"reinforcement-learning-based-heuristics-to","title":"Reinforcement Learning-based Heuristics to Guide Domain-Independent Dynamic Programming","date":"2025-03-20","arxiv_id":"2503.16371","repositories_listed":1,"syntology":null},{"url":"/paper/application-of-linear-regression-method-to","slug":"application-of-linear-regression-method-to","title":"Application of linear regression method to the deep reinforcement learning in continuous action cases","date":"2025-03-19","arxiv_id":"2503.14976","repositories_listed":1,"syntology":null},{"url":"/paper/had-gen-human-like-and-diverse-driving","slug":"had-gen-human-like-and-diverse-driving","title":"HAD-Gen: Human-like and Diverse Driving Behavior Modeling for Controllable Scenario Generation","date":"2025-03-19","arxiv_id":"2503.15049","repositories_listed":1,"syntology":null},{"url":"/paper/learning-with-expert-abstractions-for","slug":"learning-with-expert-abstractions-for","title":"Learning with Expert Abstractions for Efficient Multi-Task Continuous Control","date":"2025-03-19","arxiv_id":"2503.14809","repositories_listed":1,"syntology":null},{"url":"/paper/neural-lyapunov-function-approximation-with","slug":"neural-lyapunov-function-approximation-with","title":"Neural Lyapunov Function Approximation with Self-Supervised Reinforcement Learning","date":"2025-03-19","arxiv_id":"2503.15629","repositories_listed":1,"syntology":null},{"url":"/paper/counterfactual-experience-augmented-off","slug":"counterfactual-experience-augmented-off","title":"Counterfactual experience augmented off-policy reinforcement learning","date":"2025-03-18","arxiv_id":"2503.13842","repositories_listed":1,"syntology":null},{"url":"/paper/dapo-an-open-source-llm-reinforcement","slug":"dapo-an-open-source-llm-reinforcement","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","date":"2025-03-18","arxiv_id":"2503.14476","repositories_listed":1,"syntology":null},{"url":"/paper/socialjax-an-evaluation-suite-for-multi-agent","slug":"socialjax-an-evaluation-suite-for-multi-agent","title":"SocialJax: An Evaluation Suite for Multi-agent Reinforcement Learning in Sequential Social Dilemmas","date":"2025-03-18","arxiv_id":"2503.14576","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/socialjax-an-evaluation-suite-for-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2503.14576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.14576"}},"official":{"repos":["cooperativex/socialjax"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ipcgrl-language-instructed-reinforcement","slug":"ipcgrl-language-instructed-reinforcement","title":"IPCGRL: Language-Instructed Reinforcement Learning for Procedural Level Generation","date":"2025-03-16","arxiv_id":"2503.12358","repositories_listed":1,"syntology":null},{"url":"/paper/terl-large-scale-multi-target-encirclement","slug":"terl-large-scale-multi-target-encirclement","title":"TERL: Large-Scale Multi-Target Encirclement Using Transformer-Enhanced Reinforcement Learning","date":"2025-03-16","arxiv_id":"2503.12395","repositories_listed":1,"syntology":null},{"url":"/paper/from-demonstrations-to-rewards-alignment","slug":"from-demonstrations-to-rewards-alignment","title":"From Demonstrations to Rewards: Alignment Without Explicit Human Preferences","date":"2025-03-15","arxiv_id":"2503.13538","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/from-demonstrations-to-rewards-alignment#ran","syntology_url":"https://syntology.ai/paper/2503.13538","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.13538"}},"official":{"repos":["Hong-Lab-UMN-ECE/IRLAlignment"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-optimal-offline-reinforcement","slug":"towards-optimal-offline-reinforcement","title":"Towards Optimal Offline Reinforcement Learning","date":"2025-03-15","arxiv_id":"2503.12283","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-reset-in-target-search-problems","slug":"learning-to-reset-in-target-search-problems","title":"Learning to reset in target search problems","date":"2025-03-14","arxiv_id":"2503.11330","repositories_listed":1,"syntology":null},{"url":"/paper/rema-learning-to-meta-think-for-llms-with","slug":"rema-learning-to-meta-think-for-llms-with","title":"ReMA: Learning to Meta-think for LLMs with Multi-Agent Reinforcement Learning","date":"2025-03-12","arxiv_id":"2503.09501","repositories_listed":1,"syntology":null},{"url":"/paper/regulatory-dna-sequence-design-with","slug":"regulatory-dna-sequence-design-with","title":"Regulatory DNA sequence Design with Reinforcement Learning","date":"2025-03-11","arxiv_id":"2503.07981","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regulatory-dna-sequence-design-with#ran","syntology_url":"https://syntology.ai/paper/2503.07981","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07981"}},"official":{"repos":["yangzhao1230/taco"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/v-max-making-rl-practical-for-autonomous","slug":"v-max-making-rl-practical-for-autonomous","title":"V-Max: A Reinforcement Learning Framework for Autonomous Driving","date":"2025-03-11","arxiv_id":"2503.08388","repositories_listed":1,"syntology":null},{"url":"/paper/dad-distilled-reinforcement-learning-for","slug":"dad-distilled-reinforcement-learning-for","title":"DaD: Distilled Reinforcement Learning for Diverse Keypoint Detection","date":"2025-03-10","arxiv_id":"2503.07347","repositories_listed":1,"syntology":null},{"url":"/paper/automated-proof-of-polynomial-inequalities","slug":"automated-proof-of-polynomial-inequalities","title":"Automated Proof of Polynomial Inequalities via Reinforcement Learning","date":"2025-03-09","arxiv_id":"2503.06592","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/automated-proof-of-polynomial-inequalities#ran","syntology_url":"https://syntology.ai/paper/2503.06592","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06592"}},"official":{"repos":["blliu6/appirl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamics-invariant-quadrotor-control-using","slug":"dynamics-invariant-quadrotor-control-using","title":"Dynamics-Invariant Quadrotor Control using Scale-Aware Deep Reinforcement Learning","date":"2025-03-09","arxiv_id":"2503.09622","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-policy-optimization-for-offline","slug":"adversarial-policy-optimization-for-offline","title":"Adversarial Policy Optimization for Offline Preference-based Reinforcement Learning","date":"2025-03-07","arxiv_id":"2503.05306","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adversarial-policy-optimization-for-offline#ran","syntology_url":"https://syntology.ai/paper/2503.05306","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.05306"}},"official":{"repos":["oh-lab/APPO"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/performance-comparisons-of-reinforcement","slug":"performance-comparisons-of-reinforcement","title":"Performance Comparisons of Reinforcement Learning Algorithms for Sequential Experimental Design","date":"2025-03-07","arxiv_id":"2503.05905","repositories_listed":1,"syntology":null},{"url":"/paper/policy-constraint-by-only-support-constraint","slug":"policy-constraint-by-only-support-constraint","title":"Policy Constraint by Only Support Constraint for Offline Reinforcement Learning","date":"2025-03-07","arxiv_id":"2503.05207","repositories_listed":1,"syntology":null},{"url":"/paper/r1-omni-explainable-omni-multimodal-emotion","slug":"r1-omni-explainable-omni-multimodal-emotion","title":"R1-Omni: Explainable Omni-Multimodal Emotion Recognition with Reinforcement Learning","date":"2025-03-07","arxiv_id":"2503.05379","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"0 ran · 3 unverified","sample_list":"/paper/r1-omni-explainable-omni-multimodal-emotion#ran","syntology_url":"https://syntology.ai/paper/2503.05379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.05379"}},"official":null}},{"url":"/paper/r1-zero-s-aha-moment-in-visual-reasoning-on-a","slug":"r1-zero-s-aha-moment-in-visual-reasoning-on-a","title":"R1-Zero's \"Aha Moment\" in Visual Reasoning on a 2B Non-SFT Model","date":"2025-03-07","arxiv_id":"2503.05132","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/r1-zero-s-aha-moment-in-visual-reasoning-on-a#ran","syntology_url":"https://syntology.ai/paper/2503.05132","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.05132"}},"official":{"repos":["turningpoint-ai/visualthinker-r1-zero"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/experience-replay-with-random-reshuffling","slug":"experience-replay-with-random-reshuffling","title":"Experience Replay with Random Reshuffling","date":"2025-03-04","arxiv_id":"2503.02269","repositories_listed":1,"syntology":null},{"url":"/paper/trajectory-class-aware-multi-agent","slug":"trajectory-class-aware-multi-agent","title":"Trajectory-Class-Aware Multi-Agent Reinforcement Learning","date":"2025-03-03","arxiv_id":"2503.01440","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/trajectory-class-aware-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2503.01440","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.01440"}},"official":{"repos":["aailab-kaist/trama"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/on-generalization-across-environments-in","slug":"on-generalization-across-environments-in","title":"On Generalization Across Environments In Multi-Objective Reinforcement Learning","date":"2025-03-02","arxiv_id":"2503.00799","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-combinatorial-1","slug":"reinforcement-learning-with-combinatorial-1","title":"Reinforcement learning with combinatorial actions for coupled restless bandits","date":"2025-03-01","arxiv_id":"2503.01919","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-combinatorial-1#ran","syntology_url":"https://syntology.ai/paper/2503.01919","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.01919"}},"official":{"repos":["lily-x/combinatorial-rmab"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deepretrieval-powerful-query-generation-for","slug":"deepretrieval-powerful-query-generation-for","title":"DeepRetrieval: Hacking Real Search Engines and Retrievers with Large Language Models via Reinforcement Learning","date":"2025-02-28","arxiv_id":"2503.00223","repositories_listed":1,"syntology":{"n":25,"n_ran":24,"n_constructed":0,"n_ran_checked":23,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":23,"n_pointer_only":0,"phrase":"24 ran (of which 0 constructed an object rather than computing a result; 23 with no instrument failure: 0 honoured, 0 violated, 23 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deepretrieval-powerful-query-generation-for#ran","syntology_url":"https://syntology.ai/paper/2503.00223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00223"}},"official":{"repos":["pat-jj/deepretrieval"],"state":"official (archive's flag): 24 ran","n_ran":24,"n_constructed":0,"n_ran_no_instrument_failure":23,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/highly-parallelized-reinforcement-learning","slug":"highly-parallelized-reinforcement-learning","title":"Highly Parallelized Reinforcement Learning Training with Relaxed Assignment Dependencies","date":"2025-02-27","arxiv_id":"2502.20190","repositories_listed":1,"syntology":null},{"url":"/paper/offline-reinforcement-learning-via-inverse","slug":"offline-reinforcement-learning-via-inverse","title":"Offline Reinforcement Learning via Inverse Optimization","date":"2025-02-27","arxiv_id":"2502.20030","repositories_listed":1,"syntology":null},{"url":"/paper/playing-pokemon-red-via-deep-reinforcement","slug":"playing-pokemon-red-via-deep-reinforcement","title":"Playing Pokémon Red via Deep Reinforcement Learning","date":"2025-02-27","arxiv_id":"2502.19920","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/playing-pokemon-red-via-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2502.19920","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.19920"}},"official":{"repos":["MarcoMeter/neroRL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rize-regularized-imitation-learning-via","slug":"rize-regularized-imitation-learning-via","title":"RIZE: Regularized Imitation Learning via Distributional Reinforcement Learning","date":"2025-02-27","arxiv_id":"2502.20089","repositories_listed":1,"syntology":null},{"url":"/paper/controlling-dynamics-of-stochastic-systems","slug":"controlling-dynamics-of-stochastic-systems","title":"Controlling dynamics of stochastic systems with deep reinforcement learning","date":"2025-02-25","arxiv_id":"2502.18111","repositories_listed":1,"syntology":null},{"url":"/paper/notagen-advancing-musicality-in-symbolic","slug":"notagen-advancing-musicality-in-symbolic","title":"NotaGen: Advancing Musicality in Symbolic Music Generation with Large Language Model Training Paradigms","date":"2025-02-25","arxiv_id":"2502.18008","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/notagen-advancing-musicality-in-symbolic#ran","syntology_url":"https://syntology.ai/paper/2502.18008","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.18008"}},"official":null}},{"url":"/paper/tdmpbc-self-imitative-reinforcement-learning","slug":"tdmpbc-self-imitative-reinforcement-learning","title":"TDMPBC: Self-Imitative Reinforcement Learning for Humanoid Robot Control","date":"2025-02-24","arxiv_id":"2502.17322","repositories_listed":1,"syntology":null},{"url":"/paper/pmat-optimizing-action-generation-order-in","slug":"pmat-optimizing-action-generation-order-in","title":"PMAT: Optimizing Action Generation Order in Multi-Agent Reinforcement Learning","date":"2025-02-23","arxiv_id":"2502.16496","repositories_listed":1,"syntology":null},{"url":"/paper/risk-averse-reinforcement-learning-an-optimal","slug":"risk-averse-reinforcement-learning-an-optimal","title":"Risk-Averse Reinforcement Learning: An Optimal Transport Perspective on Temporal Difference Learning","date":"2025-02-22","arxiv_id":"2502.16328","repositories_listed":1,"syntology":null},{"url":"/paper/statistical-inference-in-reinforcement","slug":"statistical-inference-in-reinforcement","title":"Statistical Inference in Reinforcement Learning: A Selective Survey","date":"2025-02-22","arxiv_id":"2502.16195","repositories_listed":1,"syntology":null},{"url":"/paper/generating-p-functional-molecules-using-stgg","slug":"generating-p-functional-molecules-using-stgg","title":"Generating $π$-Functional Molecules Using STGG+ with Active Learning","date":"2025-02-20","arxiv_id":"2502.14842","repositories_listed":1,"syntology":null},{"url":"/paper/navigating-demand-uncertainty-in-container","slug":"navigating-demand-uncertainty-in-container","title":"Navigating Demand Uncertainty in Container Shipping: Deep Reinforcement Learning for Enabling Adaptive and Feasible Master Stowage Planning","date":"2025-02-18","arxiv_id":"2502.12756","repositories_listed":1,"syntology":null},{"url":"/paper/maximum-entropy-reinforcement-learning-with-1","slug":"maximum-entropy-reinforcement-learning-with-1","title":"Maximum Entropy Reinforcement Learning with Diffusion Policy","date":"2025-02-17","arxiv_id":"2502.11612","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":6,"n_ran_checked":8,"n_instrument":2,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":6,"n_pointer_only":13,"phrase":"10 ran (of which 6 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/maximum-entropy-reinforcement-learning-with-1#ran","syntology_url":"https://syntology.ai/paper/2502.11612","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11612"}},"official":{"repos":["diffusionyes/maxentdp"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":6,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-the-paperclip-maximizer-are-rl","slug":"evaluating-the-paperclip-maximizer-are-rl","title":"Evaluating the Paperclip Maximizer: Are RL-Based Language Models More Likely to Pursue Instrumental Goals?","date":"2025-02-16","arxiv_id":"2502.12206","repositories_listed":1,"syntology":null},{"url":"/paper/reinforced-large-language-model-is-a-formal","slug":"reinforced-large-language-model-is-a-formal","title":"Reinforced Large Language Model is a formal theorem prover","date":"2025-02-13","arxiv_id":"2502.08908","repositories_listed":1,"syntology":null},{"url":"/paper/active-advantage-aligned-online-reinforcement","slug":"active-advantage-aligned-online-reinforcement","title":"Active Advantage-Aligned Online Reinforcement Learning with Offline Data","date":"2025-02-11","arxiv_id":"2502.07937","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/active-advantage-aligned-online-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2502.07937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.07937"}},"official":{"repos":["xuefeng-cs/a3rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-constraint-violation-signals-for","slug":"leveraging-constraint-violation-signals-for","title":"Leveraging Constraint Violation Signals For Action-Constrained Reinforcement Learning","date":"2025-02-08","arxiv_id":"2502.10431","repositories_listed":1,"syntology":null},{"url":"/paper/mol-moe-training-preference-guided-routers","slug":"mol-moe-training-preference-guided-routers","title":"Mol-MoE: Training Preference-Guided Routers for Molecule Generation","date":"2025-02-08","arxiv_id":"2502.05633","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mol-moe-training-preference-guided-routers#ran","syntology_url":"https://syntology.ai/paper/2502.05633","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.05633"}},"official":{"repos":["ddidacus/mol-moe"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-convergence-and-stability-of-upside","slug":"on-the-convergence-and-stability-of-upside","title":"On the Convergence and Stability of Upside-Down Reinforcement Learning, Goal-Conditioned Supervised Learning, and Online Decision Transformers","date":"2025-02-08","arxiv_id":"2502.05672","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-graph-of-thoughts-test-time-adaptive","slug":"adaptive-graph-of-thoughts-test-time-adaptive","title":"Adaptive Graph of Thoughts: Test-Time Adaptive Reasoning Unifying Chain, Tree, and Graph Structures","date":"2025-02-07","arxiv_id":"2502.05078","repositories_listed":1,"syntology":null},{"url":"/paper/deep-meta-coordination-graphs-for-multi-agent","slug":"deep-meta-coordination-graphs-for-multi-agent","title":"Deep Meta Coordination Graphs for Multi-agent Reinforcement Learning","date":"2025-02-06","arxiv_id":"2502.04028","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-with-focal","slug":"multi-agent-reinforcement-learning-with-focal","title":"Multi-Agent Reinforcement Learning with Focal Diversity Optimization","date":"2025-02-06","arxiv_id":"2502.04492","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/multi-agent-reinforcement-learning-with-focal#ran","syntology_url":"https://syntology.ai/paper/2502.04492","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.04492"}},"official":{"repos":["sftekin/rl-focal"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/interactive-symbolic-regression-through","slug":"interactive-symbolic-regression-through","title":"Interactive Symbolic Regression through Offline Reinforcement Learning: A Co-Design Framework","date":"2025-02-05","arxiv_id":"2502.02917","repositories_listed":1,"syntology":null},{"url":"/paper/wolfpack-adversarial-attack-for-robust-multi","slug":"wolfpack-adversarial-attack-for-robust-multi","title":"Wolfpack Adversarial Attack for Robust Multi-Agent Reinforcement Learning","date":"2025-02-05","arxiv_id":"2502.02844","repositories_listed":1,"syntology":{"n":7,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/wolfpack-adversarial-attack-for-robust-multi#ran","syntology_url":"https://syntology.ai/paper/2502.02844","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.02844"}},"official":{"repos":["sunwoolee0504/wall"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/reusing-embeddings-reproducible-reward-model","slug":"reusing-embeddings-reproducible-reward-model","title":"Reusing Embeddings: Reproducible Reward Model Research in Large Language Model Alignment without GPUs","date":"2025-02-04","arxiv_id":"2502.04357","repositories_listed":1,"syntology":null},{"url":"/paper/fedhpd-heterogeneous-federated-reinforcement","slug":"fedhpd-heterogeneous-federated-reinforcement","title":"FedHPD: Heterogeneous Federated Reinforcement Learning via Policy Distillation","date":"2025-02-02","arxiv_id":"2502.00870","repositories_listed":1,"syntology":null},{"url":"/paper/sharpie-a-modular-framework-for-reinforcement","slug":"sharpie-a-modular-framework-for-reinforcement","title":"SHARPIE: A Modular Framework for Reinforcement Learning and Human-AI Interaction Experiments","date":"2025-01-31","arxiv_id":"2501.19245","repositories_listed":1,"syntology":null},{"url":"/paper/vintix-action-model-via-in-context","slug":"vintix-action-model-via-in-context","title":"Vintix: Action Model via In-Context Reinforcement Learning","date":"2025-01-31","arxiv_id":"2501.19400","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vintix-action-model-via-in-context#ran","syntology_url":"https://syntology.ai/paper/2501.19400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.19400"}},"official":{"repos":["dunnolab/vintix"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dream-to-drive-with-predictive-individual","slug":"dream-to-drive-with-predictive-individual","title":"Dream to Drive with Predictive Individual World Model","date":"2025-01-28","arxiv_id":"2501.16733","repositories_listed":1,"syntology":null},{"url":"/paper/on-rollouts-in-model-based-reinforcement","slug":"on-rollouts-in-model-based-reinforcement","title":"On Rollouts in Model-Based Reinforcement Learning","date":"2025-01-28","arxiv_id":"2501.16918","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-rollouts-in-model-based-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2501.16918","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.16918"}},"official":{"repos":["data-science-in-mechanical-engineering/infoprop"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-quantum-reinforcement-learning","slug":"benchmarking-quantum-reinforcement-learning","title":"Benchmarking Quantum Reinforcement Learning","date":"2025-01-27","arxiv_id":"2501.15893","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-quantum-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2501.15893","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.15893"}},"official":{"repos":["nicomeyer96/qrl-benchmark"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inverse-reinforcement-learning-via-convex","slug":"inverse-reinforcement-learning-via-convex","title":"Inverse Reinforcement Learning via Convex Optimization","date":"2025-01-27","arxiv_id":"2501.15957","repositories_listed":1,"syntology":null},{"url":"/paper/multi-objective-reinforcement-learning-for-2","slug":"multi-objective-reinforcement-learning-for-2","title":"Multi-Objective Reinforcement Learning for Power Grid Topology Control","date":"2025-01-27","arxiv_id":"2502.00040","repositories_listed":1,"syntology":null},{"url":"/paper/refill-reinforcement-learning-for-fill-in","slug":"refill-reinforcement-learning-for-fill-in","title":"ReFill: Reinforcement Learning for Fill-In Minimization","date":"2025-01-27","arxiv_id":"2501.16130","repositories_listed":1,"syntology":null},{"url":"/paper/upside-down-reinforcement-learning-with","slug":"upside-down-reinforcement-learning-with","title":"Upside Down Reinforcement Learning with Policy Generators","date":"2025-01-27","arxiv_id":"2501.16288","repositories_listed":1,"syntology":null},{"url":"/paper/expert-free-online-transfer-learning-in-multi-1","slug":"expert-free-online-transfer-learning-in-multi-1","title":"Expert-Free Online Transfer Learning in Multi-Agent Reinforcement Learning","date":"2025-01-26","arxiv_id":"2501.15495","repositories_listed":1,"syntology":null},{"url":"/paper/divergence-augmented-policy-optimization-1","slug":"divergence-augmented-policy-optimization-1","title":"Divergence-Augmented Policy Optimization","date":"2025-01-25","arxiv_id":"2501.15034","repositories_listed":1,"syntology":null},{"url":"/paper/improving-retrieval-augmented-generation","slug":"improving-retrieval-augmented-generation","title":"Improving Retrieval-Augmented Generation through Multi-Agent Reinforcement Learning","date":"2025-01-25","arxiv_id":"2501.15228","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-retrieval-augmented-generation#ran","syntology_url":"https://syntology.ai/paper/2501.15228","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.15228"}},"official":{"repos":["chenyiqun/mmoa-rag"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/agentrec-agent-recommendation-using-sentence","slug":"agentrec-agent-recommendation-using-sentence","title":"AgentRec: Agent Recommendation Using Sentence Embeddings Aligned to Human Feedback","date":"2025-01-23","arxiv_id":"2501.13333","repositories_listed":1,"syntology":null},{"url":"/paper/scalable-safe-multi-agent-reinforcement","slug":"scalable-safe-multi-agent-reinforcement","title":"Scalable Safe Multi-Agent Reinforcement Learning for Multi-Agent System","date":"2025-01-23","arxiv_id":"2501.13727","repositories_listed":1,"syntology":null},{"url":"/paper/utilizing-evolution-strategies-to-train","slug":"utilizing-evolution-strategies-to-train","title":"Utilizing Evolution Strategies to Train Transformers in Reinforcement Learning","date":"2025-01-23","arxiv_id":"2501.13883","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/utilizing-evolution-strategies-to-train#ran","syntology_url":"https://syntology.ai/paper/2501.13883","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.13883"}},"official":{"repos":["mafi412/evolution-strategies-and-decision-transformers"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/wfcrl-a-multi-agent-reinforcement-learning","slug":"wfcrl-a-multi-agent-reinforcement-learning","title":"WFCRL: A Multi-Agent Reinforcement Learning Benchmark for Wind Farm Control","date":"2025-01-23","arxiv_id":"2501.13592","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-data-exploitation-in-deep","slug":"adaptive-data-exploitation-in-deep","title":"Adaptive Data Exploitation in Deep Reinforcement Learning","date":"2025-01-22","arxiv_id":"2501.12620","repositories_listed":1,"syntology":null},{"url":"/paper/srmt-shared-memory-for-multi-agent-lifelong","slug":"srmt-shared-memory-for-multi-agent-lifelong","title":"SRMT: Shared Memory for Multi-agent Lifelong Pathfinding","date":"2025-01-22","arxiv_id":"2501.13200","repositories_listed":1,"syntology":null},{"url":"/paper/tackling-uncertainties-in-multi-agent","slug":"tackling-uncertainties-in-multi-agent","title":"Tackling Uncertainties in Multi-Agent Reinforcement Learning through Integration of Agent Termination Dynamics","date":"2025-01-21","arxiv_id":"2501.12061","repositories_listed":1,"syntology":null},{"url":"/paper/curiosity-driven-reinforcement-learning-from","slug":"curiosity-driven-reinforcement-learning-from","title":"Curiosity-Driven Reinforcement Learning from Human Feedback","date":"2025-01-20","arxiv_id":"2501.11463","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/curiosity-driven-reinforcement-learning-from#ran","syntology_url":"https://syntology.ai/paper/2501.11463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.11463"}},"official":{"repos":["ernie-research/cd-rlhf"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pearl-preconditioner-enhancement-through","slug":"pearl-preconditioner-enhancement-through","title":"PEARL: Preconditioner Enhancement through Actor-critic Reinforcement Learning","date":"2025-01-18","arxiv_id":"2501.10750","repositories_listed":1,"syntology":null},{"url":"/paper/ansr-dt-an-adaptive-neuro-symbolic-learning","slug":"ansr-dt-an-adaptive-neuro-symbolic-learning","title":"ANSR-DT: An Adaptive Neuro-Symbolic Learning and Reasoning Framework for Digital Twins","date":"2025-01-15","arxiv_id":"2501.08561","repositories_listed":1,"syntology":null},{"url":"/paper/cuasmrl-optimizing-gpu-sass-schedules-via","slug":"cuasmrl-optimizing-gpu-sass-schedules-via","title":"CuAsmRL: Optimizing GPU SASS Schedules via Deep Reinforcement Learning","date":"2025-01-14","arxiv_id":"2501.08071","repositories_listed":1,"syntology":null},{"url":"/paper/a-hybrid-framework-for-reinsurance","slug":"a-hybrid-framework-for-reinsurance","title":"A Hybrid Framework for Reinsurance Optimization: Integrating Generative Models and Reinforcement Learning","date":"2025-01-11","arxiv_id":"2501.06404","repositories_listed":1,"syntology":null},{"url":"/paper/the-meta-representation-hypothesis","slug":"the-meta-representation-hypothesis","title":"Representation Convergence: Mutual Distillation is Secretly a Form of Regularization","date":"2025-01-05","arxiv_id":"2501.02481","repositories_listed":1,"syntology":null},{"url":"/paper/noise-resilient-symbolic-regression-with","slug":"noise-resilient-symbolic-regression-with","title":"Noise-Resilient Symbolic Regression with Dynamic Gating Reinforcement Learning","date":"2025-01-02","arxiv_id":"2501.01085","repositories_listed":1,"syntology":null},{"url":"/paper/hybridising-reinforcement-learning-and","slug":"hybridising-reinforcement-learning-and","title":"Hybridising Reinforcement Learning and Heuristics for Hierarchical Directed Arc Routing Problems","date":"2025-01-01","arxiv_id":"2501.00852","repositories_listed":1,"syntology":null},{"url":"/paper/lease-offline-preference-based-reinforcement","slug":"lease-offline-preference-based-reinforcement","title":"LEASE: Offline Preference-based Reinforcement Learning with High Sample Efficiency","date":"2024-12-30","arxiv_id":"2412.21001","repositories_listed":1,"syntology":null},{"url":"/paper/diminishing-return-of-value-expansion-methods-1","slug":"diminishing-return-of-value-expansion-methods-1","title":"Diminishing Return of Value Expansion Methods","date":"2024-12-29","arxiv_id":"2412.20537","repositories_listed":1,"syntology":null},{"url":"/paper/numerical-solutions-of-fixed-points-in-two","slug":"numerical-solutions-of-fixed-points-in-two","title":"Numerical solutions of fixed points in two-dimensional Kuramoto-Sivashinsky equation expedited by reinforcement learning","date":"2024-12-27","arxiv_id":"2501.00046","repositories_listed":1,"syntology":null},{"url":"/paper/constraint-adaptive-policy-switching-for","slug":"constraint-adaptive-policy-switching-for","title":"Constraint-Adaptive Policy Switching for Offline Safe Reinforcement Learning","date":"2024-12-25","arxiv_id":"2412.18946","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/constraint-adaptive-policy-switching-for#ran","syntology_url":"https://syntology.ai/paper/2412.18946","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.18946"}},"official":{"repos":["yassinech/caps"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-optimization-of-portfolio-allocation","slug":"dynamic-optimization-of-portfolio-allocation","title":"A Deep Reinforcement Learning Framework for Dynamic Portfolio Optimization: Evidence from China's Stock Market","date":"2024-12-24","arxiv_id":"2412.18563","repositories_listed":1,"syntology":null},{"url":"/paper/llm-powered-user-simulator-for-recommender","slug":"llm-powered-user-simulator-for-recommender","title":"LLM-Powered User Simulator for Recommender System","date":"2024-12-22","arxiv_id":"2412.16984","repositories_listed":1,"syntology":null},{"url":"/paper/autonomous-option-invention-for-continual","slug":"autonomous-option-invention-for-continual","title":"Autonomous Option Invention for Continual Hierarchical Reinforcement Learning and Planning","date":"2024-12-20","arxiv_id":"2412.16395","repositories_listed":1,"syntology":null},{"url":"/paper/decoding-fairness-a-reinforcement-learning","slug":"decoding-fairness-a-reinforcement-learning","title":"Decoding fairness: a reinforcement learning perspective","date":"2024-12-20","arxiv_id":"2412.16249","repositories_listed":1,"syntology":null},{"url":"/paper/fedrlhf-a-convergence-guaranteed-federated","slug":"fedrlhf-a-convergence-guaranteed-federated","title":"FedRLHF: A Convergence-Guaranteed Federated Framework for Privacy-Preserving and Personalized RLHF","date":"2024-12-20","arxiv_id":"2412.15538","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-for-24","slug":"multi-agent-reinforcement-learning-for-24","title":"Multi Agent Reinforcement Learning for Sequential Satellite Assignment Problems","date":"2024-12-20","arxiv_id":"2412.15573","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-agent-reinforcement-learning-for-24#ran","syntology_url":"https://syntology.ai/paper/2412.15573","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.15573"}},"official":{"repos":["Rainlabuw/rl-enabled-distributed-assignment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-with-time-scale","slug":"deep-reinforcement-learning-with-time-scale","title":"Deep reinforcement learning with time-scale invariant memory","date":"2024-12-19","arxiv_id":"2412.15292","repositories_listed":1,"syntology":null},{"url":"/paper/offline-safe-reinforcement-learning-using","slug":"offline-safe-reinforcement-learning-using","title":"Offline Safe Reinforcement Learning Using Trajectory Classification","date":"2024-12-19","arxiv_id":"2412.15429","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/offline-safe-reinforcement-learning-using#ran","syntology_url":"https://syntology.ai/paper/2412.15429","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.15429"}},"official":{"repos":["zgong11/TraC"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/guiding-generative-protein-language-models","slug":"guiding-generative-protein-language-models","title":"Guiding Generative Protein Language Models with Reinforcement Learning","date":"2024-12-17","arxiv_id":"2412.12979","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/guiding-generative-protein-language-models#ran","syntology_url":"https://syntology.ai/paper/2412.12979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.12979"}},"official":{"repos":["ai4pdlab/dpo_plm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tilted-quantile-gradient-updates-for-quantile","slug":"tilted-quantile-gradient-updates-for-quantile","title":"Tilted Quantile Gradient Updates for Quantile-Constrained Reinforcement Learning","date":"2024-12-17","arxiv_id":"2412.13184","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-reward-design-for-reinforcement","slug":"adaptive-reward-design-for-reinforcement","title":"Adaptive Reward Design for Reinforcement Learning","date":"2024-12-14","arxiv_id":"2412.10917","repositories_listed":1,"syntology":null},{"url":"/paper/latent-safety-constrained-policy-approach-for","slug":"latent-safety-constrained-policy-approach-for","title":"Latent Safety-Constrained Policy Approach for Safe Offline Reinforcement Learning","date":"2024-12-11","arxiv_id":"2412.08794","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/latent-safety-constrained-policy-approach-for#ran","syntology_url":"https://syntology.ai/paper/2412.08794","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.08794"}},"official":{"repos":["PrajwalKoirala/LSPC-Safe-Offline-RL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"85739068be9f0cce85e3d3e0da9a2aa2fdb525693ccefd278f6fe8c835500158","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}