{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/13","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":13,"pages_in_order":132,"rows_per_page":100,"rows":[1201,1300],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/12","next":"/task/reinforcement-learning/papers/14","papers":[{"url":"/paper/a-method-for-evaluating-hyperparameter","slug":"a-method-for-evaluating-hyperparameter","title":"A Method for Evaluating Hyperparameter Sensitivity in Reinforcement Learning","date":"2024-12-10","arxiv_id":"2412.07165","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-method-for-evaluating-hyperparameter#ran","syntology_url":"https://syntology.ai/paper/2412.07165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.07165"}},"official":{"repos":["jadkins99/hyperparameter_sensitivity"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/configx-modular-configuration-for","slug":"configx-modular-configuration-for","title":"ConfigX: Modular Configuration for Evolutionary Algorithms via Multitask Reinforcement Learning","date":"2024-12-10","arxiv_id":"2412.07507","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-policy-as-macro","slug":"reinforcement-learning-policy-as-macro","title":"Reinforcement Learning Policy as Macro Regulator Rather than Macro Placer","date":"2024-12-10","arxiv_id":"2412.07167","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reinforcement-learning-policy-as-macro#ran","syntology_url":"https://syntology.ai/paper/2412.07167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.07167"}},"official":{"repos":["lamda-bbo/macro-regulator"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/off-policy-maximum-entropy-rl-with-future","slug":"off-policy-maximum-entropy-rl-with-future","title":"Off-Policy Maximum Entropy RL with Future State and Action Visitation Measures","date":"2024-12-09","arxiv_id":"2412.06655","repositories_listed":1,"syntology":null},{"url":"/paper/simudice-offline-policy-optimization-through","slug":"simudice-offline-policy-optimization-through","title":"SimuDICE: Offline Policy Optimization Through World Model Updates and DICE Estimation","date":"2024-12-09","arxiv_id":"2412.06486","repositories_listed":1,"syntology":null},{"url":"/paper/classifier-free-guidance-in-llms-safety","slug":"classifier-free-guidance-in-llms-safety","title":"Classifier-free guidance in LLMs Safety","date":"2024-12-08","arxiv_id":"2412.06846","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/classifier-free-guidance-in-llms-safety#ran","syntology_url":"https://syntology.ai/paper/2412.06846","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06846"}},"official":{"repos":["rgsmirnov/cfg_safety_llm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-an-overview","slug":"reinforcement-learning-an-overview","title":"Reinforcement Learning: An Overview","date":"2024-12-06","arxiv_id":"2412.05265","repositories_listed":1,"syntology":null},{"url":"/paper/gram-generalization-in-deep-rl-with-a-robust","slug":"gram-generalization-in-deep-rl-with-a-robust","title":"GRAM: Generalization in Deep RL with a Robust Adaptation Module","date":"2024-12-05","arxiv_id":"2412.04323","repositories_listed":1,"syntology":null},{"url":"/paper/marvel-accelerating-safe-online-reinforcement","slug":"marvel-accelerating-safe-online-reinforcement","title":"Marvel: Accelerating Safe Online Reinforcement Learning with Finetuned Offline Policy","date":"2024-12-05","arxiv_id":"2412.04426","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-enhanced-llms-a-survey","slug":"reinforcement-learning-enhanced-llms-a-survey","title":"Reinforcement Learning Enhanced LLMs: A Survey","date":"2024-12-05","arxiv_id":"2412.10400","repositories_listed":1,"syntology":null},{"url":"/paper/conformal-symplectic-optimization-for-stable","slug":"conformal-symplectic-optimization-for-stable","title":"Conformal Symplectic Optimization for Stable Reinforcement Learning","date":"2024-12-03","arxiv_id":"2412.02291","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-plastic-waste-collection-in-water","slug":"optimizing-plastic-waste-collection-in-water","title":"Optimizing Plastic Waste Collection in Water Bodies Using Heterogeneous Autonomous Surface Vehicles with Deep Reinforcement Learning","date":"2024-12-03","arxiv_id":"2412.02316","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-generative-policies-a-simpler","slug":"revisiting-generative-policies-a-simpler","title":"Revisiting Generative Policies: A Simpler Reinforcement Learning Algorithmic Perspective","date":"2024-12-02","arxiv_id":"2412.01245","repositories_listed":1,"syntology":null},{"url":"/paper/towards-fault-tolerance-in-multi-agent","slug":"towards-fault-tolerance-in-multi-agent","title":"Towards Fault Tolerance in Multi-Agent Reinforcement Learning","date":"2024-11-30","arxiv_id":"2412.00534","repositories_listed":1,"syntology":null},{"url":"/paper/carel-instruction-guided-reinforcement","slug":"carel-instruction-guided-reinforcement","title":"CAREL: Instruction-guided reinforcement learning with cross-modal auxiliary objectives","date":"2024-11-29","arxiv_id":"2411.19787","repositories_listed":1,"syntology":null},{"url":"/paper/continual-deep-reinforcement-learning-with","slug":"continual-deep-reinforcement-learning-with","title":"Continual Deep Reinforcement Learning with Task-Agnostic Policy Distillation","date":"2024-11-25","arxiv_id":"2411.16532","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/continual-deep-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2411.16532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16532"}},"official":{"repos":["wabbajack1/tapd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/creating-hierarchical-dispositions-of-needs","slug":"creating-hierarchical-dispositions-of-needs","title":"Creating Hierarchical Dispositions of Needs in an Agent","date":"2024-11-23","arxiv_id":"2412.00044","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-environments-for-vehicle-routing","slug":"multi-agent-environments-for-vehicle-routing","title":"Multi-Agent Environments for Vehicle Routing Problems","date":"2024-11-21","arxiv_id":"2411.14411","repositories_listed":1,"syntology":null},{"url":"/paper/natural-language-reinforcement-learning-1","slug":"natural-language-reinforcement-learning-1","title":"Natural Language Reinforcement Learning","date":"2024-11-21","arxiv_id":"2411.14251","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/natural-language-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2411.14251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.14251"}},"official":{"repos":["waterhorse1/natural-language-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/syllabus-portable-curricula-for-reinforcement","slug":"syllabus-portable-curricula-for-reinforcement","title":"Syllabus: Portable Curricula for Reinforcement Learning Agents","date":"2024-11-18","arxiv_id":"2411.11318","repositories_listed":1,"syntology":null},{"url":"/paper/an-investigation-of-offline-reinforcement","slug":"an-investigation-of-offline-reinforcement","title":"An Investigation of Offline Reinforcement Learning in Factorisable Action Spaces","date":"2024-11-17","arxiv_id":"2411.11088","repositories_listed":1,"syntology":null},{"url":"/paper/precision-focused-reinforcement-learning","slug":"precision-focused-reinforcement-learning","title":"Precision-Focused Reinforcement Learning Model for Robotic Object Pushing","date":"2024-11-13","arxiv_id":"2411.08622","repositories_listed":1,"syntology":null},{"url":"/paper/recommender-systems-and-reinforcement","slug":"recommender-systems-and-reinforcement","title":"Recommender systems and reinforcement learning for human-building interaction and context-aware support: A text mining-driven review of scientific literature","date":"2024-11-13","arxiv_id":"2411.08734","repositories_listed":1,"syntology":null},{"url":"/paper/doubly-mild-generalization-for-offline","slug":"doubly-mild-generalization-for-offline","title":"Doubly Mild Generalization for Offline Reinforcement Learning","date":"2024-11-12","arxiv_id":"2411.07934","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/doubly-mild-generalization-for-offline#ran","syntology_url":"https://syntology.ai/paper/2411.07934","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.07934"}},"official":{"repos":["maoyixiu/dmg"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/non-adversarial-inverse-reinforcement","slug":"non-adversarial-inverse-reinforcement","title":"Non-Adversarial Inverse Reinforcement Learning via Successor Feature Matching","date":"2024-11-11","arxiv_id":"2411.07007","repositories_listed":1,"syntology":{"n":24,"n_ran":19,"n_constructed":14,"n_ran_checked":15,"n_instrument":4,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":14,"n_pointer_only":0,"phrase":"19 ran (of which 14 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 1 violated, 14 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/non-adversarial-inverse-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2411.07007","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.07007"}},"official":{"repos":["arnavkj1995/sfm"],"state":"official (archive's flag): 19 ran","n_ran":19,"n_constructed":14,"n_ran_no_instrument_failure":15,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-quantum-tiq-taq","slug":"reinforcement-learning-for-quantum-tiq-taq","title":"Reinforcement learning for Quantum Tiq-Taq-Toe","date":"2024-11-10","arxiv_id":"2411.06429","repositories_listed":1,"syntology":null},{"url":"/paper/state-chrono-representation-for-enhancing","slug":"state-chrono-representation-for-enhancing","title":"State Chrono Representation for Enhancing Generalization in Reinforcement Learning","date":"2024-11-09","arxiv_id":"2411.06174","repositories_listed":1,"syntology":null},{"url":"/paper/tangled-program-graphs-as-an-alternative-to","slug":"tangled-program-graphs-as-an-alternative-to","title":"Tangled Program Graphs as an alternative to DRL-based control algorithms for UAVs","date":"2024-11-08","arxiv_id":"2411.05586","repositories_listed":1,"syntology":null},{"url":"/paper/constrained-latent-action-policies-for-model","slug":"constrained-latent-action-policies-for-model","title":"Constrained Latent Action Policies for Model-Based Offline Reinforcement Learning","date":"2024-11-07","arxiv_id":"2411.04562","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/constrained-latent-action-policies-for-model#ran","syntology_url":"https://syntology.ai/paper/2411.04562","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.04562"}},"official":{"repos":["marvinalles/c-lap"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/hypercube-policy-regularization-framework-for","slug":"hypercube-policy-regularization-framework-for","title":"Hypercube Policy Regularization Framework for Offline Reinforcement Learning","date":"2024-11-07","arxiv_id":"2411.04534","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-aware-resource-management-for-c-v2x","slug":"semantic-aware-resource-management-for-c-v2x","title":"Semantic-Aware Resource Management for C-V2X Platooning via Multi-Agent Reinforcement Learning","date":"2024-11-07","arxiv_id":"2411.04672","repositories_listed":1,"syntology":null},{"url":"/paper/think-smart-act-smarl-analyzing-probabilistic","slug":"think-smart-act-smarl-analyzing-probabilistic","title":"Think Smart, Act SMARL! Analyzing Probabilistic Logic Shields for Multi-Agent Reinforcement Learning","date":"2024-11-07","arxiv_id":"2411.04867","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-assist-humans-without-inferring","slug":"learning-to-assist-humans-without-inferring","title":"Learning to Assist Humans without Inferring Rewards","date":"2024-11-04","arxiv_id":"2411.02623","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/learning-to-assist-humans-without-inferring#ran","syntology_url":"https://syntology.ai/paper/2411.02623","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02623"}},"official":{"repos":["vivekmyers/empowerment_successor_representations"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/enhancing-chess-reinforcement-learning-with","slug":"enhancing-chess-reinforcement-learning-with","title":"Enhancing Chess Reinforcement Learning with Graph Representation","date":"2024-10-31","arxiv_id":"2410.23753","repositories_listed":1,"syntology":null},{"url":"/paper/from-easy-to-hard-tackling-quantum-problems","slug":"from-easy-to-hard-tackling-quantum-problems","title":"Reinforcement learning with learned gadgets to tackle hard quantum problems on real hardware","date":"2024-10-31","arxiv_id":"2411.00230","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/from-easy-to-hard-tackling-quantum-problems#ran","syntology_url":"https://syntology.ai/paper/2411.00230","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00230"}},"official":{"repos":["aqasch/gadget_rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/prosody-as-a-teaching-signal-for-agent","slug":"prosody-as-a-teaching-signal-for-agent","title":"Prosody as a Teaching Signal for Agent Learning: Exploratory Studies and Algorithmic Implications","date":"2024-10-31","arxiv_id":"2410.23554","repositories_listed":1,"syntology":null},{"url":"/paper/ra-pbrl-provably-efficient-risk-aware","slug":"ra-pbrl-provably-efficient-risk-aware","title":"RA-PbRL: Provably Efficient Risk-Aware Preference-Based Reinforcement Learning","date":"2024-10-31","arxiv_id":"2410.23569","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-gradients-as-vitamin","slug":"reinforcement-learning-gradients-as-vitamin","title":"Reinforcement Learning Gradients as Vitamin for Online Finetuning Decision Transformers","date":"2024-10-31","arxiv_id":"2410.24108","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reinforcement-learning-gradients-as-vitamin#ran","syntology_url":"https://syntology.ai/paper/2410.24108","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.24108"}},"official":{"repos":["kaiyan289/rl_as_vitamin_for_online_decision_transformers"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/econojax-a-fast-scalable-economic-simulation","slug":"econojax-a-fast-scalable-economic-simulation","title":"EconoJax: A Fast & Scalable Economic Simulation in Jax","date":"2024-10-29","arxiv_id":"2410.22165","repositories_listed":1,"syntology":null},{"url":"/paper/predicting-future-actions-of-reinforcement","slug":"predicting-future-actions-of-reinforcement","title":"Predicting Future Actions of Reinforcement Learning Agents","date":"2024-10-29","arxiv_id":"2410.22459","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/predicting-future-actions-of-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2410.22459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22459"}},"official":{"repos":["stephen-chung-mh/predict_action"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fairstream-fair-multimedia-streaming","slug":"fairstream-fair-multimedia-streaming","title":"FairStream: Fair Multimedia Streaming Benchmark for Reinforcement Learning Agents","date":"2024-10-28","arxiv_id":"2410.21029","repositories_listed":1,"syntology":null},{"url":"/paper/falcon-feedback-driven-adaptive-long-short","slug":"falcon-feedback-driven-adaptive-long-short","title":"FALCON: Feedback-driven Adaptive Long/short-term memory reinforced Coding Optimization system","date":"2024-10-28","arxiv_id":"2410.21349","repositories_listed":1,"syntology":null},{"url":"/paper/odrl-a-benchmark-for-off-dynamics","slug":"odrl-a-benchmark-for-off-dynamics","title":"ODRL: A Benchmark for Off-Dynamics Reinforcement Learning","date":"2024-10-28","arxiv_id":"2410.20750","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/odrl-a-benchmark-for-off-dynamics#ran","syntology_url":"https://syntology.ai/paper/2410.20750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20750"}},"official":{"repos":["offdynamicsrl/off-dynamics-rl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-modeling-with-weak-supervision-for","slug":"reward-modeling-with-weak-supervision-for","title":"Reward Modeling with Weak Supervision for Language Models","date":"2024-10-28","arxiv_id":"2410.20869","repositories_listed":1,"syntology":null},{"url":"/paper/robustness-and-generalization-in-quantum","slug":"robustness-and-generalization-in-quantum","title":"Robustness and Generalization in Quantum Reinforcement Learning via Lipschitz Regularization","date":"2024-10-28","arxiv_id":"2410.21117","repositories_listed":1,"syntology":null},{"url":"/paper/ogbench-benchmarking-offline-goal-conditioned","slug":"ogbench-benchmarking-offline-goal-conditioned","title":"OGBench: Benchmarking Offline Goal-Conditioned RL","date":"2024-10-26","arxiv_id":"2410.20092","repositories_listed":1,"syntology":null},{"url":"/paper/toward-finding-strong-pareto-optimal-policies","slug":"toward-finding-strong-pareto-optimal-policies","title":"Toward Finding Strong Pareto Optimal Policies in Multi-Agent Reinforcement Learning","date":"2024-10-25","arxiv_id":"2410.19372","repositories_listed":1,"syntology":null},{"url":"/paper/entity-based-reinforcement-learning-for","slug":"entity-based-reinforcement-learning-for","title":"Entity-based Reinforcement Learning for Autonomous Cyber Defence","date":"2024-10-23","arxiv_id":"2410.17647","repositories_listed":1,"syntology":null},{"url":"/paper/scalable-offline-reinforcement-learning-for","slug":"scalable-offline-reinforcement-learning-for","title":"Scalable Offline Reinforcement Learning for Mean Field Games","date":"2024-10-23","arxiv_id":"2410.17898","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-two-player-performance-through","slug":"enhancing-two-player-performance-through","title":"Enhancing Two-Player Performance Through Single-Player Knowledge Transfer: An Empirical Study on Atari 2600 Games","date":"2024-10-22","arxiv_id":"2410.16653","repositories_listed":1,"syntology":null},{"url":"/paper/evolution-with-opponent-learning-awareness","slug":"evolution-with-opponent-learning-awareness","title":"Evolution of Societies via Reinforcement Learning","date":"2024-10-22","arxiv_id":"2410.17466","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-rl-based-llm-training-for-formal","slug":"exploring-rl-based-llm-training-for-formal","title":"Exploring RL-based LLM Training for Formal Language Tasks with Programmed Rewards","date":"2024-10-22","arxiv_id":"2410.17126","repositories_listed":1,"syntology":null},{"url":"/paper/navigating-noisy-feedback-enhancing","slug":"navigating-noisy-feedback-enhancing","title":"Navigating Noisy Feedback: Enhancing Reinforcement Learning with Error-Prone Language Models","date":"2024-10-22","arxiv_id":"2410.17389","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/navigating-noisy-feedback-enhancing#ran","syntology_url":"https://syntology.ai/paper/2410.17389","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17389"}},"official":{"repos":["sy-shi/RLAIF_ScoreDiff"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforced-imitative-trajectory-planning-for","slug":"reinforced-imitative-trajectory-planning-for","title":"Reinforced Imitative Trajectory Planning for Urban Automated Driving","date":"2024-10-21","arxiv_id":"2410.15607","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-dynamic-memory","slug":"reinforcement-learning-for-dynamic-memory","title":"Reinforcement Learning for Dynamic Memory Allocation","date":"2024-10-20","arxiv_id":"2410.15492","repositories_listed":1,"syntology":null},{"url":"/paper/cooperation-and-fairness-in-multi-agent","slug":"cooperation-and-fairness-in-multi-agent","title":"Cooperation and Fairness in Multi-Agent Reinforcement Learning","date":"2024-10-19","arxiv_id":"2410.14916","repositories_listed":1,"syntology":null},{"url":"/paper/intersectionzoo-eco-driving-for-benchmarking","slug":"intersectionzoo-eco-driving-for-benchmarking","title":"IntersectionZoo: Eco-driving for Benchmarking Multi-Agent Contextual Reinforcement Learning","date":"2024-10-19","arxiv_id":"2410.15221","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/intersectionzoo-eco-driving-for-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2410.15221","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.15221"}},"official":{"repos":["mit-wu-lab/IntersectionZoo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-deep-reinforcement-learning-for-1","slug":"benchmarking-deep-reinforcement-learning-for-1","title":"Benchmarking Deep Reinforcement Learning for Navigation in Denied Sensor Environments","date":"2024-10-18","arxiv_id":"2410.14616","repositories_listed":1,"syntology":null},{"url":"/paper/streaming-deep-reinforcement-learning-finally","slug":"streaming-deep-reinforcement-learning-finally","title":"Streaming Deep Reinforcement Learning Finally Works","date":"2024-10-18","arxiv_id":"2410.14606","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/streaming-deep-reinforcement-learning-finally#ran","syntology_url":"https://syntology.ai/paper/2410.14606","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14606"}},"official":{"repos":["mohmdelsayed/streaming-drl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-online-5","slug":"deep-reinforcement-learning-for-online-5","title":"Deep Reinforcement Learning for Online Optimal Execution Strategies","date":"2024-10-17","arxiv_id":"2410.13493","repositories_listed":1,"syntology":null},{"url":"/paper/bayes-adaptive-monte-carlo-tree-search-for","slug":"bayes-adaptive-monte-carlo-tree-search-for","title":"Bayes Adaptive Monte Carlo Tree Search for Offline Model-based Reinforcement Learning","date":"2024-10-15","arxiv_id":"2410.11234","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bayes-adaptive-monte-carlo-tree-search-for#ran","syntology_url":"https://syntology.ai/paper/2410.11234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11234"}},"official":{"repos":["lucascjysdl/offline-rl-kit"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-shot-model-based-reinforcement-learning","slug":"zero-shot-model-based-reinforcement-learning","title":"Zero-shot Model-based Reinforcement Learning using Large Language Models","date":"2024-10-15","arxiv_id":"2410.11711","repositories_listed":1,"syntology":null},{"url":"/paper/stable-hadamard-memory-revitalizing-memory","slug":"stable-hadamard-memory-revitalizing-memory","title":"Stable Hadamard Memory: Revitalizing Memory-Augmented Agents for Reinforcement Learning","date":"2024-10-14","arxiv_id":"2410.10132","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stable-hadamard-memory-revitalizing-memory#ran","syntology_url":"https://syntology.ai/paper/2410.10132","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10132"}},"official":null}},{"url":"/paper/improving-generalization-on-the-procgen","slug":"improving-generalization-on-the-procgen","title":"Improving Generalization on the ProcGen Benchmark with Simple Architectural Changes and Scale","date":"2024-10-13","arxiv_id":"2410.10905","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-generalization-on-the-procgen#ran","syntology_url":"https://syntology.ai/paper/2410.10905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10905"}},"official":{"repos":["anndvision/vsop-3d"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/openr-an-open-source-framework-for-advanced","slug":"openr-an-open-source-framework-for-advanced","title":"OpenR: An Open Source Framework for Advanced Reasoning with Large Language Models","date":"2024-10-12","arxiv_id":"2410.09671","repositories_listed":1,"syntology":{"n":10,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/openr-an-open-source-framework-for-advanced#ran","syntology_url":"https://syntology.ai/paper/2410.09671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09671"}},"official":null}},{"url":"/paper/top-erl-transformer-based-off-policy-episodic","slug":"top-erl-transformer-based-off-policy-episodic","title":"TOP-ERL: Transformer-based Off-Policy Episodic Reinforcement Learning","date":"2024-10-12","arxiv_id":"2410.09536","repositories_listed":1,"syntology":null},{"url":"/paper/kaleidoscope-learnable-masks-for","slug":"kaleidoscope-learnable-masks-for","title":"Kaleidoscope: Learnable Masks for Heterogeneous Multi-agent Reinforcement Learning","date":"2024-10-11","arxiv_id":"2410.08540","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":2,"n_ran_checked":8,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"11 ran (of which 2 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/kaleidoscope-learnable-masks-for#ran","syntology_url":"https://syntology.ai/paper/2410.08540","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08540"}},"official":{"repos":["lxxxxr/kaleidoscope"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":2,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"url":"/paper/learning-to-walk-from-three-minutes-of-real","slug":"learning-to-walk-from-three-minutes-of-real","title":"Learning to Walk from Three Minutes of Real-World Data with Semi-structured Dynamics Models","date":"2024-10-11","arxiv_id":"2410.09163","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-control-of-non","slug":"reinforcement-learning-for-control-of-non","title":"Reinforcement Learning for Control of Non-Markovian Cellular Population Dynamics","date":"2024-10-11","arxiv_id":"2410.08439","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-for-control-of-non#ran","syntology_url":"https://syntology.ai/paper/2410.08439","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08439"}},"official":{"repos":["JacobHA/RL4Dosing"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-natural-language-based-strategies","slug":"exploring-natural-language-based-strategies","title":"Exploring Natural Language-Based Strategies for Efficient Number Learning in Children through Reinforcement Learning","date":"2024-10-10","arxiv_id":"2410.08334","repositories_listed":1,"syntology":null},{"url":"/paper/optima-optimized-policy-for-intelligent-multi","slug":"optima-optimized-policy-for-intelligent-multi","title":"OPTIMA: Optimized Policy for Intelligent Multi-Agent Systems Enables Coordination-Aware Autonomous Vehicles","date":"2024-10-09","arxiv_id":"2410.18112","repositories_listed":1,"syntology":null},{"url":"/paper/coevolving-with-the-other-you-fine-tuning-llm","slug":"coevolving-with-the-other-you-fine-tuning-llm","title":"Coevolving with the Other You: Fine-Tuning LLM with Sequential Cooperative Multi-Agent Reinforcement Learning","date":"2024-10-08","arxiv_id":"2410.06101","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/coevolving-with-the-other-you-fine-tuning-llm#ran","syntology_url":"https://syntology.ai/paper/2410.06101","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.06101"}},"official":{"repos":["Harry67Hu/CORY"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/optimizing-the-training-schedule-of","slug":"optimizing-the-training-schedule-of","title":"Optimizing the Training Schedule of Multilingual NMT using Reinforcement Learning","date":"2024-10-08","arxiv_id":"2410.06118","repositories_listed":1,"syntology":null},{"url":"/paper/improved-off-policy-reinforcement-learning-in","slug":"improved-off-policy-reinforcement-learning-in","title":"Improved Off-policy Reinforcement Learning in Biological Sequence Design","date":"2024-10-06","arxiv_id":"2410.04461","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improved-off-policy-reinforcement-learning-in#ran","syntology_url":"https://syntology.ai/paper/2410.04461","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04461"}},"official":{"repos":["hyeonahkimm/delta_cs"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/mitigating-adversarial-perturbations-for-deep","slug":"mitigating-adversarial-perturbations-for-deep","title":"Mitigating Adversarial Perturbations for Deep Reinforcement Learning via Vector Quantization","date":"2024-10-04","arxiv_id":"2410.03376","repositories_listed":1,"syntology":null},{"url":"/paper/open-world-reinforcement-learning-over-long","slug":"open-world-reinforcement-learning-over-long","title":"Open-World Reinforcement Learning over Long Short-Term Imagination","date":"2024-10-04","arxiv_id":"2410.03618","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/open-world-reinforcement-learning-over-long#ran","syntology_url":"https://syntology.ai/paper/2410.03618","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03618"}},"official":{"repos":["qiwang067/LS-Imagine"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ma-rlhf-reinforcement-learning-from-human","slug":"ma-rlhf-reinforcement-learning-from-human","title":"MA-RLHF: Reinforcement Learning from Human Feedback with Macro Actions","date":"2024-10-03","arxiv_id":"2410.02743","repositories_listed":1,"syntology":null},{"url":"/paper/maniskill3-gpu-parallelized-robotics","slug":"maniskill3-gpu-parallelized-robotics","title":"ManiSkill3: GPU Parallelized Robotics Simulation and Rendering for Generalizable Embodied AI","date":"2024-10-01","arxiv_id":"2410.00425","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/maniskill3-gpu-parallelized-robotics#ran","syntology_url":"https://syntology.ai/paper/2410.00425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00425"}},"official":{"repos":["haosulab/ManiSkill"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inferring-preferences-from-demonstrations-in-2","slug":"inferring-preferences-from-demonstrations-in-2","title":"Inferring Preferences from Demonstrations in Multi-objective Reinforcement Learning","date":"2024-09-30","arxiv_id":"2409.20258","repositories_listed":1,"syntology":null},{"url":"/paper/constrained-reinforcement-learning-for-safe","slug":"constrained-reinforcement-learning-for-safe","title":"Constrained Reinforcement Learning for Safe Heat Pump Control","date":"2024-09-29","arxiv_id":"2409.19716","repositories_listed":1,"syntology":null},{"url":"/paper/autonomous-network-defence-using","slug":"autonomous-network-defence-using","title":"Autonomous Network Defence using Reinforcement Learning","date":"2024-09-26","arxiv_id":"2409.18197","repositories_listed":1,"syntology":null},{"url":"/paper/mathdsl-a-domain-specific-language-for","slug":"mathdsl-a-domain-specific-language-for","title":"MathDSL: A Domain-Specific Language for Concise Mathematical Solutions Via Program Synthesis","date":"2024-09-26","arxiv_id":"2409.17490","repositories_listed":1,"syntology":null},{"url":"/paper/multi-robot-informative-path-planning-for","slug":"multi-robot-informative-path-planning-for","title":"Scalable Multi-Robot Informative Path Planning for Target Mapping via Deep Reinforcement Learning","date":"2024-09-25","arxiv_id":"2409.16967","repositories_listed":1,"syntology":null},{"url":"/paper/fedslate-a-federated-deep-reinforcement","slug":"fedslate-a-federated-deep-reinforcement","title":"FedSlate:A Federated Deep Reinforcement Learning Recommender System","date":"2024-09-23","arxiv_id":"2409.14872","repositories_listed":1,"syntology":null},{"url":"/paper/exploiting-exogenous-structure-for-sample","slug":"exploiting-exogenous-structure-for-sample","title":"Exploiting Exogenous Structure for Sample-Efficient Reinforcement Learning","date":"2024-09-22","arxiv_id":"2409.14557","repositories_listed":1,"syntology":null},{"url":"/paper/handling-long-term-safety-and-uncertainty-in","slug":"handling-long-term-safety-and-uncertainty-in","title":"Handling Long-Term Safety and Uncertainty in Safe Reinforcement Learning","date":"2024-09-18","arxiv_id":"2409.12045","repositories_listed":1,"syntology":null},{"url":"/paper/secure-control-systems-for-autonomous","slug":"secure-control-systems-for-autonomous","title":"Secure Control Systems for Autonomous Quadrotors against Cyber-Attacks","date":"2024-09-18","arxiv_id":"2409.11897","repositories_listed":1,"syntology":null},{"url":"/paper/integrating-reinforcement-learning-and-model","slug":"integrating-reinforcement-learning-and-model","title":"Integrating Reinforcement Learning and Model Predictive Control with Applications to Microgrids","date":"2024-09-17","arxiv_id":"2409.11267","repositories_listed":1,"syntology":null},{"url":"/paper/audio-driven-reinforcement-learning-for-head","slug":"audio-driven-reinforcement-learning-for-head","title":"Audio-Driven Reinforcement Learning for Head-Orientation in Naturalistic Environments","date":"2024-09-16","arxiv_id":"2409.10048","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-rl-safety-with-counterfactual-llm","slug":"enhancing-rl-safety-with-counterfactual-llm","title":"Enhancing RL Safety with Counterfactual LLM Reasoning","date":"2024-09-16","arxiv_id":"2409.10188","repositories_listed":1,"syntology":null},{"url":"/paper/offline-reinforcement-learning-for-learning","slug":"offline-reinforcement-learning-for-learning","title":"Offline Reinforcement Learning for Learning to Dispatch for Job Shop Scheduling","date":"2024-09-16","arxiv_id":"2409.10589","repositories_listed":1,"syntology":null},{"url":"/paper/quantile-regression-for-distributional-reward","slug":"quantile-regression-for-distributional-reward","title":"Quantile Regression for Distributional Reward Models in RLHF","date":"2024-09-16","arxiv_id":"2409.10164","repositories_listed":1,"syntology":null},{"url":"/paper/robust-reinforcement-learning-with-dynamic","slug":"robust-reinforcement-learning-with-dynamic","title":"Robust Reinforcement Learning with Dynamic Distortion Risk Measures","date":"2024-09-16","arxiv_id":"2409.10096","repositories_listed":1,"syntology":null},{"url":"/paper/learning-discrete-world-models-for-heuristic","slug":"learning-discrete-world-models-for-heuristic","title":"Learning Discrete World Models for Heuristic Search","date":"2024-09-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/design-optimization-of-nuclear-fusion-reactor","slug":"design-optimization-of-nuclear-fusion-reactor","title":"Design Optimization of Nuclear Fusion Reactor through Deep Reinforcement Learning","date":"2024-09-12","arxiv_id":"2409.08231","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-discovers-efficient","slug":"reinforcement-learning-discovers-efficient","title":"Reinforcement Learning Discovers Efficient Decentralized Graph Path Search Strategies","date":"2024-09-12","arxiv_id":"2409.07932","repositories_listed":1,"syntology":null},{"url":"/paper/one-policy-to-run-them-all-an-end-to-end","slug":"one-policy-to-run-them-all-an-end-to-end","title":"One Policy to Run Them All: an End-to-end Learning Approach to Multi-Embodiment Locomotion","date":"2024-09-10","arxiv_id":"2409.06366","repositories_listed":1,"syntology":null},{"url":"/paper/combating-spatial-disorientation-in-a-dynamic","slug":"combating-spatial-disorientation-in-a-dynamic","title":"Combating Spatial Disorientation in a Dynamic Self-Stabilization Task Using AI Assistants","date":"2024-09-09","arxiv_id":"2409.14565","repositories_listed":1,"syntology":null},{"url":"/paper/semifactual-explanations-for-reinforcement","slug":"semifactual-explanations-for-reinforcement","title":"Semifactual Explanations for Reinforcement Learning","date":"2024-09-09","arxiv_id":"2409.05435","repositories_listed":1,"syntology":null},{"url":"/paper/elo-rated-sequence-rewards-advancing","slug":"elo-rated-sequence-rewards-advancing","title":"ELO-Rated Sequence Rewards: Advancing Reinforcement Learning Models","date":"2024-09-05","arxiv_id":"2409.03301","repositories_listed":1,"syntology":null}],"record_sha256":"552363975ff6481c0f0e32307eadfbc01e4e9b9af4c36c31bbcf0228f0fd0d6d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}