{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/18","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":18,"pages_in_order":152,"rows_per_page":100,"rows":[1701,1800],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/17","next":"/task/reinforcement-learning-1/papers/19","papers":[{"url":"/paper/health-text-simplification-an-annotated","slug":"health-text-simplification-an-annotated","title":"Health Text Simplification: An Annotated Corpus for Digestive Cancer Education and Novel Strategies for Reinforcement Learning","date":"2024-01-26","arxiv_id":"2401.15043","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-hidden-markov","slug":"reinforcement-learning-with-hidden-markov","title":"HMM for Discovering Decision-Making Dynamics Using Reinforcement Learning Experiments","date":"2024-01-25","arxiv_id":"2401.13929","repositories_listed":1,"syntology":null},{"url":"/paper/true-knowledge-comes-from-practice-aligning","slug":"true-knowledge-comes-from-practice-aligning","title":"True Knowledge Comes from Practice: Aligning LLMs with Embodied Environments via Reinforcement Learning","date":"2024-01-25","arxiv_id":"2401.14151","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/true-knowledge-comes-from-practice-aligning#ran","syntology_url":"https://syntology.ai/paper/2401.14151","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.14151"}},"official":{"repos":["weihaotan/twosome"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/seer-facilitating-structured-reasoning-and","slug":"seer-facilitating-structured-reasoning-and","title":"SEER: Facilitating Structured Reasoning and Explanation via Reinforcement Learning","date":"2024-01-24","arxiv_id":"2401.13246","repositories_listed":1,"syntology":{"n":16,"n_ran":13,"n_constructed":0,"n_ran_checked":8,"n_instrument":5,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":2,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/seer-facilitating-structured-reasoning-and#ran","syntology_url":"https://syntology.ai/paper/2401.13246","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13246"}},"official":{"repos":["chen-gx/seer"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/hazard-challenge-embodied-decision-making-in","slug":"hazard-challenge-embodied-decision-making-in","title":"HAZARD Challenge: Embodied Decision Making in Dynamically Changing Environments","date":"2024-01-23","arxiv_id":"2401.12975","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hazard-challenge-embodied-decision-making-in#ran","syntology_url":"https://syntology.ai/paper/2401.12975","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.12975"}},"official":{"repos":["umass-foundation-model/hazard"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-evolutionary-algorithms-and","slug":"bridging-evolutionary-algorithms-and","title":"Bridging Evolutionary Algorithms and Reinforcement Learning: A Comprehensive Survey on Hybrid Algorithms","date":"2024-01-22","arxiv_id":"2401.11963","repositories_listed":1,"syntology":null},{"url":"/paper/emergent-dominance-hierarchies-in","slug":"emergent-dominance-hierarchies-in","title":"Emergent Dominance Hierarchies in Reinforcement Learning Agents","date":"2024-01-21","arxiv_id":"2401.12258","repositories_listed":1,"syntology":null},{"url":"/paper/information-theoretic-state-variable","slug":"information-theoretic-state-variable","title":"Information-Theoretic State Variable Selection for Reinforcement Learning","date":"2024-01-21","arxiv_id":"2401.11512","repositories_listed":1,"syntology":null},{"url":"/paper/open-the-black-box-step-based-policy-updates","slug":"open-the-black-box-step-based-policy-updates","title":"Open the Black Box: Step-based Policy Updates for Temporally-Correlated Episodic Reinforcement Learning","date":"2024-01-21","arxiv_id":"2401.11437","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/open-the-black-box-step-based-policy-updates#ran","syntology_url":"https://syntology.ai/paper/2401.11437","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.11437"}},"official":{"repos":["brucegeli/tce_rl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reframing-offline-reinforcement-learning-as-a","slug":"reframing-offline-reinforcement-learning-as-a","title":"Solving Offline Reinforcement Learning with Decision Tree Regression","date":"2024-01-21","arxiv_id":"2401.11630","repositories_listed":1,"syntology":null},{"url":"/paper/vqc-based-reinforcement-learning-with-data-re","slug":"vqc-based-reinforcement-learning-with-data-re","title":"VQC-Based Reinforcement Learning with Data Re-uploading: Performance and Trainability","date":"2024-01-21","arxiv_id":"2401.11555","repositories_listed":1,"syntology":null},{"url":"/paper/closing-the-gap-between-td-learning-and","slug":"closing-the-gap-between-td-learning-and","title":"Closing the Gap between TD Learning and Supervised Learning -- A Generalisation Point of View","date":"2024-01-20","arxiv_id":"2401.11237","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/closing-the-gap-between-td-learning-and#ran","syntology_url":"https://syntology.ai/paper/2401.11237","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.11237"}},"official":{"repos":["rajghugare19/stitching-is-combinatorial-generalisation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/freed-improving-rl-agents-for-fragment-based","slug":"freed-improving-rl-agents-for-fragment-based","title":"FREED++: Improving RL Agents for Fragment-Based Molecule Generation by Thorough Reproduction","date":"2024-01-18","arxiv_id":"2401.09840","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/freed-improving-rl-agents-for-fragment-based#ran","syntology_url":"https://syntology.ai/paper/2401.09840","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09840"}},"official":{"repos":["airi-institute/ffreed"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/blackout-mitigation-via-physics-guided-rl","slug":"blackout-mitigation-via-physics-guided-rl","title":"Blackout Mitigation via Physics-guided RL","date":"2024-01-17","arxiv_id":"2401.09640","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-state-and-history-representations","slug":"bridging-state-and-history-representations","title":"Bridging State and History Representations: Understanding Self-Predictive RL","date":"2024-01-17","arxiv_id":"2401.08898","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-state-and-history-representations#ran","syntology_url":"https://syntology.ai/paper/2401.08898","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.08898"}},"official":{"repos":["twni2016/self-predictive-rl"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/delf-designing-learning-environments-with","slug":"delf-designing-learning-environments-with","title":"DeLF: Designing Learning Environments with Foundation Models","date":"2024-01-17","arxiv_id":"2401.08936","repositories_listed":1,"syntology":null},{"url":"/paper/deployable-reinforcement-learning-with","slug":"deployable-reinforcement-learning-with","title":"Deployable Reinforcement Learning with Variable Control Rate","date":"2024-01-17","arxiv_id":"2401.09286","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deployable-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2401.09286","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09286"}},"official":{"repos":["alpaficia/SEAC_Pytorch_release"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/uoep-user-oriented-exploration-policy-for","slug":"uoep-user-oriented-exploration-policy-for","title":"UOEP: User-Oriented Exploration Policy for Enhancing Long-Term User Experiences in Recommender Systems","date":"2024-01-17","arxiv_id":"2401.09034","repositories_listed":1,"syntology":null},{"url":"/paper/learning-from-sparse-offline-datasets-via","slug":"learning-from-sparse-offline-datasets-via","title":"Learning from Sparse Offline Datasets via Conservative Density Estimation","date":"2024-01-16","arxiv_id":"2401.08819","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-from-sparse-offline-datasets-via#ran","syntology_url":"https://syntology.ai/paper/2401.08819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.08819"}},"official":{"repos":["czp16/cde-offline-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-large-language-models-via-fine","slug":"improving-large-language-models-via-fine","title":"Improving Large Language Models via Fine-grained Reinforcement Learning with Minimum Editing Constraint","date":"2024-01-11","arxiv_id":"2401.06081","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-large-language-models-via-fine#ran","syntology_url":"https://syntology.ai/paper/2401.06081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.06081"}},"official":{"repos":["rucaibox/rlmec"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/interpretable-concept-bottlenecks-to-align","slug":"interpretable-concept-bottlenecks-to-align","title":"Interpretable Concept Bottlenecks to Align Reinforcement Learning Agents","date":"2024-01-11","arxiv_id":"2401.05821","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/interpretable-concept-bottlenecks-to-align#ran","syntology_url":"https://syntology.ai/paper/2401.05821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05821"}},"official":{"repos":["k4ntz/scobots"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-distributional-reward-critic-architecture","slug":"the-distributional-reward-critic-architecture","title":"The Distributional Reward Critic Framework for Reinforcement Learning Under Perturbed Rewards","date":"2024-01-11","arxiv_id":"2401.05710","repositories_listed":1,"syntology":null},{"url":"/paper/using-reinforcement-learning-to-improve-drone","slug":"using-reinforcement-learning-to-improve-drone","title":"Using reinforcement learning to improve drone-based inference of greenhouse gas fluxes","date":"2024-01-08","arxiv_id":"2401.03932","repositories_listed":1,"syntology":null},{"url":"/paper/a-robust-quantile-huber-loss-with","slug":"a-robust-quantile-huber-loss-with","title":"A Robust Quantile Huber Loss With Interpretable Parameter Adjustment In Distributional Reinforcement Learning","date":"2024-01-04","arxiv_id":"2401.02325","repositories_listed":1,"syntology":null},{"url":"/paper/data-assimilation-in-chaotic-systems-using","slug":"data-assimilation-in-chaotic-systems-using","title":"Data Assimilation in Chaotic Systems Using Deep Reinforcement Learning","date":"2024-01-01","arxiv_id":"2401.00916","repositories_listed":1,"syntology":null},{"url":"/paper/dmr-decomposed-multi-modality-representations","slug":"dmr-decomposed-multi-modality-representations","title":"DMR: Decomposed Multi-Modality Representations for Frames and Events Fusion in Visual Reinforcement Learning","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/improving-unsupervised-hierarchical","slug":"improving-unsupervised-hierarchical","title":"Improving Unsupervised Hierarchical Representation with Reinforcement Learning","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/poce-primal-policy-optimization-with","slug":"poce-primal-policy-optimization-with","title":"POCE: Primal Policy Optimization with Conservative Estimation for Multi-constraint Offline Reinforcement Learning","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/online-symbolic-music-alignment-with-offline","slug":"online-symbolic-music-alignment-with-offline","title":"Online Symbolic Music Alignment with Offline Reinforcement Learning","date":"2023-12-31","arxiv_id":"2401.00466","repositories_listed":1,"syntology":null},{"url":"/paper/causal-state-distillation-for-explainable","slug":"causal-state-distillation-for-explainable","title":"Causal State Distillation for Explainable Reinforcement Learning","date":"2023-12-30","arxiv_id":"2401.00104","repositories_listed":1,"syntology":null},{"url":"/paper/laboratory-experiments-of-model-based","slug":"laboratory-experiments-of-model-based","title":"Laboratory Experiments of Model-based Reinforcement Learning for Adaptive Optics Control","date":"2023-12-30","arxiv_id":"2401.00242","repositories_listed":1,"syntology":null},{"url":"/paper/generalizable-visual-reinforcement-learning","slug":"generalizable-visual-reinforcement-learning","title":"Generalizable Visual Reinforcement Learning with Segment Anything Model","date":"2023-12-28","arxiv_id":"2312.17116","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/generalizable-visual-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2312.17116","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.17116"}},"official":{"repos":["wadiuvatzy/sam-g"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-model-based-policy-based-and-value","slug":"rethinking-model-based-policy-based-and-value","title":"Rethinking Model-based, Policy-based, and Value-based Reinforcement Learning via the Lens of Representation Complexity","date":"2023-12-28","arxiv_id":"2312.17248","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/rethinking-model-based-policy-based-and-value#ran","syntology_url":"https://syntology.ai/paper/2312.17248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.17248"}},"official":{"repos":["guhfeng/rl-representation-complexity"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/large-language-models-as-traffic-signal","slug":"large-language-models-as-traffic-signal","title":"LLMLight: Large Language Models as Traffic Signal Control Agents","date":"2023-12-26","arxiv_id":"2312.16044","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-as-traffic-signal#ran","syntology_url":"https://syntology.ai/paper/2312.16044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.16044"}},"official":{"repos":["usail-hkust/llmtscs"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/optimistic-and-pessimistic-actor-in-rl","slug":"optimistic-and-pessimistic-actor-in-rl","title":"Efficient Reinforcement Learning via Decoupling Exploration and Utilization","date":"2023-12-26","arxiv_id":"2312.15965","repositories_listed":1,"syntology":null},{"url":"/paper/critic-guided-decision-transformer-for","slug":"critic-guided-decision-transformer-for","title":"Critic-Guided Decision Transformer for Offline Reinforcement Learning","date":"2023-12-21","arxiv_id":"2312.13716","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/critic-guided-decision-transformer-for#ran","syntology_url":"https://syntology.ai/paper/2312.13716","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.13716"}},"official":{"repos":["sharkwyf/cgdt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-reward-learning-rewards-via","slug":"diffusion-reward-learning-rewards-via","title":"Diffusion Reward: Learning Rewards via Conditional Video Diffusion","date":"2023-12-21","arxiv_id":"2312.14134","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diffusion-reward-learning-rewards-via#ran","syntology_url":"https://syntology.ai/paper/2312.14134","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14134"}},"official":{"repos":["TEA-Lab/diffusion_reward"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/openrl-a-unified-reinforcement-learning","slug":"openrl-a-unified-reinforcement-learning","title":"OpenRL: A Unified Reinforcement Learning Framework","date":"2023-12-20","arxiv_id":"2312.16189","repositories_listed":1,"syntology":null},{"url":"/paper/parameterized-projected-bellman-operator","slug":"parameterized-projected-bellman-operator","title":"Parameterized Projected Bellman Operator","date":"2023-12-20","arxiv_id":"2312.12869","repositories_listed":1,"syntology":null},{"url":"/paper/rfrl-gym-a-reinforcement-learning-testbed-for","slug":"rfrl-gym-a-reinforcement-learning-testbed-for","title":"RFRL Gym: A Reinforcement Learning Testbed for Cognitive Radio Applications","date":"2023-12-20","arxiv_id":"2401.05406","repositories_listed":1,"syntology":null},{"url":"/paper/badrl-sparse-targeted-backdoor-attack-against","slug":"badrl-sparse-targeted-backdoor-attack-against","title":"BadRL: Sparse Targeted Backdoor Attack Against Reinforcement Learning","date":"2023-12-19","arxiv_id":"2312.12585","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/badrl-sparse-targeted-backdoor-attack-against#ran","syntology_url":"https://syntology.ai/paper/2312.12585","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.12585"}},"official":{"repos":["7777777cc/code"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/challenges-for-reinforcement-learning-in-1","slug":"challenges-for-reinforcement-learning-in-1","title":"Challenges for Reinforcement Learning in Quantum Circuit Design","date":"2023-12-18","arxiv_id":"2312.11337","repositories_listed":1,"syntology":null},{"url":"/paper/cacto-sl-using-sobolev-learning-to-improve","slug":"cacto-sl-using-sobolev-learning-to-improve","title":"CACTO-SL: Using Sobolev Learning to improve Continuous Actor-Critic with Trajectory Optimization","date":"2023-12-17","arxiv_id":"2312.10666","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-act-without-actions","slug":"learning-to-act-without-actions","title":"Learning to Act without Actions","date":"2023-12-17","arxiv_id":"2312.10812","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/learning-to-act-without-actions#ran","syntology_url":"https://syntology.ai/paper/2312.10812","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.10812"}},"official":{"repos":["schmidtdominik/lapo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/imitate-the-good-and-avoid-the-bad-an","slug":"imitate-the-good-and-avoid-the-bad-an","title":"Imitate the Good and Avoid the Bad: An Incremental Approach to Safe Reinforcement Learning","date":"2023-12-16","arxiv_id":"2312.10385","repositories_listed":1,"syntology":null},{"url":"/paper/improving-environment-robustness-of-deep","slug":"improving-environment-robustness-of-deep","title":"Improving Environment Robustness of Deep Reinforcement Learning Approaches for Autonomous Racing Using Bayesian Optimization-based Curriculum Learning","date":"2023-12-16","arxiv_id":"2312.10557","repositories_listed":1,"syntology":null},{"url":"/paper/the-effective-horizon-explains-deep-rl","slug":"the-effective-horizon-explains-deep-rl","title":"The Effective Horizon Explains Deep RL Performance in Stochastic Environments","date":"2023-12-13","arxiv_id":"2312.08369","repositories_listed":1,"syntology":null},{"url":"/paper/world-models-via-policy-guided-trajectory","slug":"world-models-via-policy-guided-trajectory","title":"World Models via Policy-Guided Trajectory Diffusion","date":"2023-12-13","arxiv_id":"2312.08533","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":2,"n_no_contract":2,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 2 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/world-models-via-policy-guided-trajectory#ran","syntology_url":"https://syntology.ai/paper/2312.08533","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08533"}},"official":{"repos":["marc-rigter/polygrad-world-models"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sequential-planning-in-large-partially","slug":"sequential-planning-in-large-partially","title":"Sequential Planning in Large Partially Observable Environments guided by LLMs","date":"2023-12-12","arxiv_id":"2312.07368","repositories_listed":1,"syntology":null},{"url":"/paper/traffic-signal-control-using-lightweight","slug":"traffic-signal-control-using-lightweight","title":"Traffic Signal Control Using Lightweight Transformers: An Offline-to-Online RL Approach","date":"2023-12-12","arxiv_id":"2312.07795","repositories_listed":1,"syntology":null},{"url":"/paper/reward-certification-for-policy-smoothed","slug":"reward-certification-for-policy-smoothed","title":"Reward Certification for Policy Smoothed Reinforcement Learning","date":"2023-12-11","arxiv_id":"2312.06436","repositories_listed":1,"syntology":{"n":11,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":11,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/reward-certification-for-policy-smoothed#ran","syntology_url":"https://syntology.ai/paper/2312.06436","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06436"}},"official":{"repos":["trustai/receps"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-sparse-reward-goal-conditioned","slug":"efficient-sparse-reward-goal-conditioned","title":"Efficient Sparse-Reward Goal-Conditioned Reinforcement Learning with a High Replay Ratio and Regularization","date":"2023-12-10","arxiv_id":"2312.05787","repositories_listed":1,"syntology":null},{"url":"/paper/the-generalization-gap-in-offline","slug":"the-generalization-gap-in-offline","title":"The Generalization Gap in Offline Reinforcement Learning","date":"2023-12-10","arxiv_id":"2312.05742","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/the-generalization-gap-in-offline#ran","syntology_url":"https://syntology.ai/paper/2312.05742","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.05742"}},"official":{"repos":["facebookresearch/gen_dgrl"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-calibration-of-compartmental","slug":"on-the-calibration-of-compartmental","title":"On the calibration of compartmental epidemiological models","date":"2023-12-09","arxiv_id":"2312.05456","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-parity-challenges-in-reinforcement","slug":"exploring-parity-challenges-in-reinforcement","title":"Exploring Parity Challenges in Reinforcement Learning through Curriculum Learning with Noisy Labels","date":"2023-12-08","arxiv_id":"2312.05379","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-via-1","slug":"multi-agent-reinforcement-learning-via-1","title":"Multi-Agent Reinforcement Learning via Distributed MPC as a Function Approximator","date":"2023-12-08","arxiv_id":"2312.05166","repositories_listed":1,"syntology":null},{"url":"/paper/unitsa-a-universal-reinforcement-learning","slug":"unitsa-a-universal-reinforcement-learning","title":"UniTSA: A Universal Reinforcement Learning Framework for V2X Traffic Signal Control","date":"2023-12-08","arxiv_id":"2312.05090","repositories_listed":1,"syntology":null},{"url":"/paper/codex-a-cluster-based-method-for-explainable","slug":"codex-a-cluster-based-method-for-explainable","title":"CODEX: A Cluster-Based Method for Explainable Reinforcement Learning","date":"2023-12-07","arxiv_id":"2312.04216","repositories_listed":1,"syntology":null},{"url":"/paper/is-feedback-all-you-need-leveraging-natural","slug":"is-feedback-all-you-need-leveraging-natural","title":"Is Feedback All You Need? Leveraging Natural Language Feedback in Goal-Conditioned Reinforcement Learning","date":"2023-12-07","arxiv_id":"2312.04736","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/is-feedback-all-you-need-leveraging-natural#ran","syntology_url":"https://syntology.ai/paper/2312.04736","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.04736"}},"official":{"repos":["uoe-agents/feedback-dt"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/micro-model-based-offline-reinforcement","slug":"micro-model-based-offline-reinforcement","title":"MICRO: Model-Based Offline Reinforcement Learning with a Conservative Bellman Operator","date":"2023-12-07","arxiv_id":"2312.03991","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-distributed-reinforcement-learning","slug":"optimizing-distributed-reinforcement-learning","title":"Efficient Parallel Reinforcement Learning Framework using the Reactor Model","date":"2023-12-07","arxiv_id":"2312.04704","repositories_listed":1,"syntology":null},{"url":"/paper/language-model-alignment-with-elastic-reset-1","slug":"language-model-alignment-with-elastic-reset-1","title":"Language Model Alignment with Elastic Reset","date":"2023-12-06","arxiv_id":"2312.07551","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-model-alignment-with-elastic-reset-1#ran","syntology_url":"https://syntology.ai/paper/2312.07551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.07551"}},"official":{"repos":["mnoukhov/elastic-reset"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mocha-multi-objective-reinforcement","slug":"mocha-multi-objective-reinforcement","title":"Mitigating Open-Vocabulary Caption Hallucinations","date":"2023-12-06","arxiv_id":"2312.03631","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":15,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mocha-multi-objective-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2312.03631","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03631"}},"official":{"repos":["assafbk/mocha_code"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/pearl-a-production-ready-reinforcement","slug":"pearl-a-production-ready-reinforcement","title":"Pearl: A Production-ready Reinforcement Learning Agent","date":"2023-12-06","arxiv_id":"2312.03814","repositories_listed":1,"syntology":null},{"url":"/paper/lexci-a-framework-for-reinforcement-learning","slug":"lexci-a-framework-for-reinforcement-learning","title":"LExCI: A Framework for Reinforcement Learning with Embedded Systems","date":"2023-12-05","arxiv_id":"2312.02739","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarl-benchmarking-multi-agent","slug":"benchmarl-benchmarking-multi-agent","title":"BenchMARL: Benchmarking Multi-Agent Reinforcement Learning","date":"2023-12-03","arxiv_id":"2312.01472","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarl-benchmarking-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2312.01472","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.01472"}},"official":{"repos":["facebookresearch/benchmarl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-curricula-in-open-ended-worlds","slug":"learning-curricula-in-open-ended-worlds","title":"Learning Curricula in Open-Ended Worlds","date":"2023-12-03","arxiv_id":"2312.03126","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-curricula-in-open-ended-worlds#ran","syntology_url":"https://syntology.ai/paper/2312.03126","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03126"}},"official":{"repos":["facebookresearch/dcd"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ddxt-deep-generative-transformer-models-for","slug":"ddxt-deep-generative-transformer-models-for","title":"DDxT: Deep Generative Transformer Models for Differential Diagnosis","date":"2023-12-02","arxiv_id":"2312.01242","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":5,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 5 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ddxt-deep-generative-transformer-models-for#ran","syntology_url":"https://syntology.ai/paper/2312.01242","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.01242"}},"official":{"repos":["MahmudulAlam/Differential-Diagnosis-Using-Transformers"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":5,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/harnessing-discrete-representations-for","slug":"harnessing-discrete-representations-for","title":"Harnessing Discrete Representations For Continual Reinforcement Learning","date":"2023-12-02","arxiv_id":"2312.01203","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"0 ran · 3 unverified","sample_list":"/paper/harnessing-discrete-representations-for#ran","syntology_url":"https://syntology.ai/paper/2312.01203","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.01203"}},"official":{"repos":["ejmejm/discrete-representations-for-continual-rl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/age-based-scheduling-for-mobile-edge","slug":"age-based-scheduling-for-mobile-edge","title":"Age-Based Scheduling for Mobile Edge Computing: A Deep Reinforcement Learning Approach","date":"2023-12-01","arxiv_id":"2312.00279","repositories_listed":1,"syntology":null},{"url":"/paper/tracking-object-positions-in-reinforcement","slug":"tracking-object-positions-in-reinforcement","title":"Tracking Object Positions in Reinforcement Learning: A Metric for Keypoint Detection (extended version)","date":"2023-12-01","arxiv_id":"2312.00592","repositories_listed":1,"syntology":null},{"url":"/paper/controlgym-large-scale-safety-critical","slug":"controlgym-large-scale-safety-critical","title":"Controlgym: Large-Scale Control Environments for Benchmarking Reinforcement Learning Algorithms","date":"2023-11-30","arxiv_id":"2311.18736","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-attack-and-defense-for-reinforcement","slug":"optimal-attack-and-defense-for-reinforcement","title":"Optimal Attack and Defense for Reinforcement Learning","date":"2023-11-30","arxiv_id":"2312.00198","repositories_listed":1,"syntology":null},{"url":"/paper/predictable-reinforcement-learning-dynamics","slug":"predictable-reinforcement-learning-dynamics","title":"Predictable Reinforcement Learning Dynamics through Entropy Rate Minimization","date":"2023-11-30","arxiv_id":"2311.18703","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-replaces-supervision-query","slug":"reinforcement-replaces-supervision-query","title":"Reinforcement Replaces Supervision: Query focused Summarization using Deep Reinforcement Learning","date":"2023-11-29","arxiv_id":"2311.17514","repositories_listed":1,"syntology":null},{"url":"/paper/unveiling-the-implicit-toxicity-in-large","slug":"unveiling-the-implicit-toxicity-in-large","title":"Unveiling the Implicit Toxicity in Large Language Models","date":"2023-11-29","arxiv_id":"2311.17391","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/unveiling-the-implicit-toxicity-in-large#ran","syntology_url":"https://syntology.ai/paper/2311.17391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17391"}},"official":{"repos":["thu-coai/implicit-toxicity"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/an-investigation-of-time-reversal-symmetry-in","slug":"an-investigation-of-time-reversal-symmetry-in","title":"An Investigation of Time Reversal Symmetry in Reinforcement Learning","date":"2023-11-28","arxiv_id":"2311.17008","repositories_listed":1,"syntology":null},{"url":"/paper/two-step-dynamic-obstacle-avoidance","slug":"two-step-dynamic-obstacle-avoidance","title":"Two-step dynamic obstacle avoidance","date":"2023-11-28","arxiv_id":"2311.16841","repositories_listed":1,"syntology":null},{"url":"/paper/generative-modelling-of-stochastic-actions-1","slug":"generative-modelling-of-stochastic-actions-1","title":"Generative Modelling of Stochastic Actions with Arbitrary Constraints in Reinforcement Learning","date":"2023-11-26","arxiv_id":"2311.15341","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generative-modelling-of-stochastic-actions-1#ran","syntology_url":"https://syntology.ai/paper/2311.15341","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.15341"}},"official":{"repos":["cameron-chen/flow-iar"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/margin-trader-a-reinforcement-learning","slug":"margin-trader-a-reinforcement-learning","title":"Margin Trader: A Reinforcement Learning Framework for Portfolio Management with Margin and Constraints","date":"2023-11-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/from-images-to-connections-can-dqn-with-gnns","slug":"from-images-to-connections-can-dqn-with-gnns","title":"From Images to Connections: Can DQN with GNNs learn the Strategic Game of Hex?","date":"2023-11-22","arxiv_id":"2311.13414","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-model-is-a-good-policy-teacher","slug":"large-language-model-is-a-good-policy-teacher","title":"Large Language Model as a Policy Teacher for Training Reinforcement Learning Agents","date":"2023-11-22","arxiv_id":"2311.13373","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-maskable-stock","slug":"reinforcement-learning-with-maskable-stock","title":"Reinforcement Learning with Maskable Stock Representation for Portfolio Management in Customizable Stock Pools","date":"2023-11-17","arxiv_id":"2311.10801","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-model-predictive","slug":"reinforcement-learning-with-model-predictive","title":"Reinforcement Learning with Model Predictive Control for Highway Ramp Metering","date":"2023-11-15","arxiv_id":"2311.08820","repositories_listed":1,"syntology":null},{"url":"/paper/direct-preference-optimization-for-neural","slug":"direct-preference-optimization-for-neural","title":"Direct Preference Optimization for Neural Machine Translation with Minimum Bayes Risk Decoding","date":"2023-11-14","arxiv_id":"2311.08380","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/direct-preference-optimization-for-neural#ran","syntology_url":"https://syntology.ai/paper/2311.08380","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08380"}},"official":{"repos":["bruceyg/dpo-mbr"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/combinatorial-optimization-with-policy","slug":"combinatorial-optimization-with-policy","title":"Combinatorial Optimization with Policy Adaptation using Latent Space Search","date":"2023-11-13","arxiv_id":"2311.13569","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/combinatorial-optimization-with-policy#ran","syntology_url":"https://syntology.ai/paper/2311.13569","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.13569"}},"official":{"repos":["instadeepai/compass"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-solving-stochastic","slug":"reinforcement-learning-for-solving-stochastic","title":"Reinforcement Learning for Solving Stochastic Vehicle Routing Problem","date":"2023-11-13","arxiv_id":"2311.07708","repositories_listed":1,"syntology":null},{"url":"/paper/clipped-objective-policy-gradients-for","slug":"clipped-objective-policy-gradients-for","title":"Clipped-Objective Policy Gradients for Pessimistic Policy Optimization","date":"2023-11-10","arxiv_id":"2311.05846","repositories_listed":1,"syntology":null},{"url":"/paper/accelerating-exploration-with-unlabeled-prior-1","slug":"accelerating-exploration-with-unlabeled-prior-1","title":"Accelerating Exploration with Unlabeled Prior Data","date":"2023-11-09","arxiv_id":"2311.05067","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":2,"n_ran_checked":9,"n_instrument":3,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":14,"phrase":"12 ran (of which 2 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/accelerating-exploration-with-unlabeled-prior-1#ran","syntology_url":"https://syntology.ai/paper/2311.05067","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.05067"}},"official":{"repos":["facebookresearch/explore"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/llm-augmented-hierarchical-agents","slug":"llm-augmented-hierarchical-agents","title":"LLM Augmented Hierarchical Agents","date":"2023-11-09","arxiv_id":"2311.05596","repositories_listed":1,"syntology":null},{"url":"/paper/uni-o4-unifying-online-and-offline-deep","slug":"uni-o4-unifying-online-and-offline-deep","title":"Uni-O4: Unifying Online and Offline Deep Reinforcement Learning with Multi-Step On-Policy Optimization","date":"2023-11-06","arxiv_id":"2311.03351","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uni-o4-unifying-online-and-offline-deep#ran","syntology_url":"https://syntology.ai/paper/2311.03351","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.03351"}},"official":{"repos":["Lei-Kun/Uni-O4"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/alberdice-addressing-out-of-distribution-1","slug":"alberdice-addressing-out-of-distribution-1","title":"AlberDICE: Addressing Out-Of-Distribution Joint Actions in Offline Multi-Agent RL via Alternating Stationary Distribution Correction Estimation","date":"2023-11-03","arxiv_id":"2311.02194","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/alberdice-addressing-out-of-distribution-1#ran","syntology_url":"https://syntology.ai/paper/2311.02194","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.02194"}},"official":{"repos":["dematsunaga/alberdice"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchical-reinforcement-learning-for-power","slug":"hierarchical-reinforcement-learning-for-power","title":"Hierarchical Reinforcement Learning for Power Network Topology Control","date":"2023-11-03","arxiv_id":"2311.02129","repositories_listed":1,"syntology":null},{"url":"/paper/state-wise-safe-reinforcement-learning-with","slug":"state-wise-safe-reinforcement-learning-with","title":"State-Wise Safe Reinforcement Learning With Pixel Observations","date":"2023-11-03","arxiv_id":"2311.02227","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":14,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/state-wise-safe-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2311.02227","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.02227"}},"official":{"repos":["simonzhan-code/step-wise_saferl_pixel"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-models-for-reinforcement-learning-a","slug":"diffusion-models-for-reinforcement-learning-a","title":"Diffusion Models for Reinforcement Learning: A Survey","date":"2023-11-02","arxiv_id":"2311.01223","repositories_listed":1,"syntology":null},{"url":"/paper/unleashing-the-power-of-pre-trained-language","slug":"unleashing-the-power-of-pre-trained-language","title":"Unleashing the Power of Pre-trained Language Models for Offline Reinforcement Learning","date":"2023-10-31","arxiv_id":"2310.20587","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unleashing-the-power-of-pre-trained-language#ran","syntology_url":"https://syntology.ai/paper/2310.20587","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20587"}},"official":{"repos":["srzer/LaMo-2023"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/variational-curriculum-reinforcement-learning","slug":"variational-curriculum-reinforcement-learning","title":"Variational Curriculum Reinforcement Learning for Unsupervised Discovery of Skills","date":"2023-10-30","arxiv_id":"2310.19424","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/variational-curriculum-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2310.19424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19424"}},"official":{"repos":["seongun-kim/vcrl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmark-generation-framework-with","slug":"benchmark-generation-framework-with","title":"Benchmark Generation Framework with Customizable Distortions for Image Classifier Robustness","date":"2023-10-28","arxiv_id":"2310.18626","repositories_listed":1,"syntology":null},{"url":"/paper/robust-offline-policy-evaluation-and","slug":"robust-offline-policy-evaluation-and","title":"Robust Offline Reinforcement learning with Heavy-Tailed Rewards","date":"2023-10-28","arxiv_id":"2310.18715","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-offline-policy-evaluation-and#ran","syntology_url":"https://syntology.ai/paper/2310.18715","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.18715"}},"official":{"repos":["mamba413/room"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-distributionally-robust-learning-and","slug":"bridging-distributionally-robust-learning-and","title":"Bridging Distributionally Robust Learning and Offline RL: An Approach to Mitigate Distribution Shift and Partial Data Coverage","date":"2023-10-27","arxiv_id":"2310.18434","repositories_listed":1,"syntology":null}],"record_sha256":"c3c03bc146a102b1204fbfc5e465dc1d8e63f5be2b0f532ab4bbace3b3778a25","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}