{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/18","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":18,"pages_in_order":132,"rows_per_page":100,"rows":[1701,1800],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/17","next":"/task/reinforcement-learning/papers/19","papers":[{"url":"/paper/application-of-deep-and-reinforcement","slug":"application-of-deep-and-reinforcement","title":"Application of deep and reinforcement learning to boundary control problems","date":"2023-10-21","arxiv_id":"2310.15191","repositories_listed":1,"syntology":null},{"url":"/paper/stabilizing-reinforcement-learning-control-a","slug":"stabilizing-reinforcement-learning-control-a","title":"Stabilizing reinforcement learning control: A modular framework for optimizing over all stable behavior","date":"2023-10-21","arxiv_id":"2310.14098","repositories_listed":1,"syntology":null},{"url":"/paper/rl-x-a-deep-reinforcement-learning-library","slug":"rl-x-a-deep-reinforcement-learning-library","title":"RL-X: A Deep Reinforcement Learning Library (not only) for RoboCup","date":"2023-10-20","arxiv_id":"2310.13396","repositories_listed":1,"syntology":null},{"url":"/paper/safe-rlhf-safe-reinforcement-learning-from","slug":"safe-rlhf-safe-reinforcement-learning-from","title":"Safe RLHF: Safe Reinforcement Learning from Human Feedback","date":"2023-10-19","arxiv_id":"2310.12773","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safe-rlhf-safe-reinforcement-learning-from#ran","syntology_url":"https://syntology.ai/paper/2310.12773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12773"}},"official":{"repos":["pku-alignment/safe-rlhf"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/quality-diversity-through-human-feedback","slug":"quality-diversity-through-human-feedback","title":"Quality Diversity through Human Feedback: Towards Open-Ended Diversity-Driven Optimization","date":"2023-10-18","arxiv_id":"2310.12103","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quality-diversity-through-human-feedback#ran","syntology_url":"https://syntology.ai/paper/2310.12103","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12103"}},"official":{"repos":["ld-ing/qdhf"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/non-ergodicity-in-reinforcement-learning","slug":"non-ergodicity-in-reinforcement-learning","title":"Reinforcement learning with non-ergodic reward increments: robustness via ergodicity transformations","date":"2023-10-17","arxiv_id":"2310.11335","repositories_listed":1,"syntology":null},{"url":"/paper/personalized-soups-personalized-large","slug":"personalized-soups-personalized-large","title":"Personalized Soups: Personalized Large Language Model Alignment via Post-hoc Parameter Merging","date":"2023-10-17","arxiv_id":"2310.11564","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":15,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/personalized-soups-personalized-large#ran","syntology_url":"https://syntology.ai/paper/2310.11564","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.11564"}},"official":{"repos":["joeljang/rlphf"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/building-persona-consistent-dialogue-agents","slug":"building-persona-consistent-dialogue-agents","title":"Building Persona Consistent Dialogue Agents with Offline Reinforcement Learning","date":"2023-10-16","arxiv_id":"2310.10735","repositories_listed":1,"syntology":null},{"url":"/paper/machine-learning-in-physics-a-short-guide","slug":"machine-learning-in-physics-a-short-guide","title":"Machine learning in physics: a short guide","date":"2023-10-16","arxiv_id":"2310.10368","repositories_listed":1,"syntology":null},{"url":"/paper/a-partially-supervised-reinforcement-learning","slug":"a-partially-supervised-reinforcement-learning","title":"A Partially Supervised Reinforcement Learning Framework for Visual Active Search","date":"2023-10-15","arxiv_id":"2310.09689","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-partially-supervised-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2310.09689","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09689"}},"official":{"repos":["anindyasarkariith/psrl_vas"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/amago-scalable-in-context-reinforcement","slug":"amago-scalable-in-context-reinforcement","title":"AMAGO: Scalable In-Context Reinforcement Learning for Adaptive Agents","date":"2023-10-15","arxiv_id":"2310.09971","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/amago-scalable-in-context-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2310.09971","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09971"}},"official":{"repos":["ut-austin-rpl/amago"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/specialized-deep-residual-policy-safe","slug":"specialized-deep-residual-policy-safe","title":"Specialized Deep Residual Policy Safe Reinforcement Learning-Based Controller for Complex and Continuous State-Action Spaces","date":"2023-10-15","arxiv_id":"2310.14788","repositories_listed":1,"syntology":null},{"url":"/paper/storm-efficient-stochastic-transformer-based-1","slug":"storm-efficient-stochastic-transformer-based-1","title":"STORM: Efficient Stochastic Transformer based World Models for Reinforcement Learning","date":"2023-10-14","arxiv_id":"2310.09615","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/storm-efficient-stochastic-transformer-based-1#ran","syntology_url":"https://syntology.ai/paper/2310.09615","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09615"}},"official":{"repos":["weipu-zhang/storm"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/transformers-as-decision-makers-provable-in","slug":"transformers-as-decision-makers-provable-in","title":"Transformers as Decision Makers: Provable In-Context Reinforcement Learning via Supervised Pretraining","date":"2023-10-12","arxiv_id":"2310.08566","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/transformers-as-decision-makers-provable-in#ran","syntology_url":"https://syntology.ai/paper/2310.08566","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08566"}},"official":{"repos":["licong-lin/in-context-rl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/virtual-augmented-reality-for-atari","slug":"virtual-augmented-reality-for-atari","title":"Virtual Augmented Reality for Atari Reinforcement Learning","date":"2023-10-12","arxiv_id":"2310.08683","repositories_listed":1,"syntology":null},{"url":"/paper/imitation-learning-from-observation-with","slug":"imitation-learning-from-observation-with","title":"Imitation Learning from Observation with Automatic Discount Scheduling","date":"2023-10-11","arxiv_id":"2310.07433","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":1,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 1 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/imitation-learning-from-observation-with#ran","syntology_url":"https://syntology.ai/paper/2310.07433","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.07433"}},"official":null}},{"url":"/paper/roboclip-one-demonstration-is-enough-to-learn","slug":"roboclip-one-demonstration-is-enough-to-learn","title":"RoboCLIP: One Demonstration is Enough to Learn Robot Policies","date":"2023-10-11","arxiv_id":"2310.07899","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/roboclip-one-demonstration-is-enough-to-learn#ran","syntology_url":"https://syntology.ai/paper/2310.07899","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.07899"}},"official":null}},{"url":"/paper/boosting-continuous-control-with-consistency","slug":"boosting-continuous-control-with-consistency","title":"Boosting Continuous Control with Consistency Policy","date":"2023-10-10","arxiv_id":"2310.06343","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/boosting-continuous-control-with-consistency#ran","syntology_url":"https://syntology.ai/paper/2310.06343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.06343"}},"official":{"repos":["cccedric/cpql"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-uncovers","slug":"deep-reinforcement-learning-uncovers","title":"Deep reinforcement learning uncovers processes for separating azeotropic mixtures without prior knowledge","date":"2023-10-10","arxiv_id":"2310.06415","repositories_listed":1,"syntology":null},{"url":"/paper/safe-deep-policy-adaptation","slug":"safe-deep-policy-adaptation","title":"Safe Deep Policy Adaptation","date":"2023-10-08","arxiv_id":"2310.08602","repositories_listed":1,"syntology":null},{"url":"/paper/surgical-gym-a-high-performance-gpu-based","slug":"surgical-gym-a-high-performance-gpu-based","title":"Surgical Gym: A high-performance GPU-based platform for reinforcement learning with surgical robots","date":"2023-10-07","arxiv_id":"2310.04676","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-uniform-sampling-offline-reinforcement-1","slug":"beyond-uniform-sampling-offline-reinforcement-1","title":"Beyond Uniform Sampling: Offline Reinforcement Learning with Imbalanced Datasets","date":"2023-10-06","arxiv_id":"2310.04413","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/beyond-uniform-sampling-offline-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2310.04413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04413"}},"official":{"repos":["Improbable-AI/dw-offline-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/drift-deep-reinforcement-learning-for-1","slug":"drift-deep-reinforcement-learning-for-1","title":"DRIFT: Deep Reinforcement Learning for Intelligent Floating Platforms Trajectories","date":"2023-10-06","arxiv_id":"2310.04266","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-neuron-segmentation-with","slug":"self-supervised-neuron-segmentation-with","title":"Self-Supervised Neuron Segmentation with Multi-Agent Reinforcement Learning","date":"2023-10-06","arxiv_id":"2310.04148","repositories_listed":1,"syntology":{"n":24,"n_ran":20,"n_constructed":4,"n_ran_checked":18,"n_instrument":2,"n_unverified":4,"n_honours":2,"n_violates":1,"n_no_contract":15,"n_pointer_only":24,"phrase":"20 ran (of which 4 constructed an object rather than computing a result; 18 with no instrument failure: 2 honoured, 1 violated, 15 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/self-supervised-neuron-segmentation-with#ran","syntology_url":"https://syntology.ai/paper/2310.04148","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04148"}},"official":{"repos":["ydchen0806/dbmim"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":4,"n_ran_no_instrument_failure":18,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/lesson-learning-to-integrate-exploration","slug":"lesson-learning-to-integrate-exploration","title":"LESSON: Learning to Integrate Exploration Strategies for Reinforcement Learning via an Option Framework","date":"2023-10-05","arxiv_id":"2310.03342","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lesson-learning-to-integrate-exploration#ran","syntology_url":"https://syntology.ai/paper/2310.03342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03342"}},"official":{"repos":["beanie00/lesson"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/discovering-general-reinforcement-learning-1","slug":"discovering-general-reinforcement-learning-1","title":"Discovering General Reinforcement Learning Algorithms with Adversarial Environment Design","date":"2023-10-04","arxiv_id":"2310.02782","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/discovering-general-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2310.02782","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02782"}},"official":{"repos":["EmptyJackson/groove"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-reinforcement-learning-for-power","slug":"multi-agent-reinforcement-learning-for-power","title":"Multi-Agent Reinforcement Learning for Power Grid Topology Optimization","date":"2023-10-04","arxiv_id":"2310.02605","repositories_listed":1,"syntology":null},{"url":"/paper/differentially-encoded-observation-spaces-for","slug":"differentially-encoded-observation-spaces-for","title":"Differentially Encoded Observation Spaces for Perceptive Reinforcement Learning","date":"2023-10-03","arxiv_id":"2310.01767","repositories_listed":1,"syntology":null},{"url":"/paper/prioritized-soft-q-decomposition-for","slug":"prioritized-soft-q-decomposition-for","title":"Prioritized Soft Q-Decomposition for Lexicographic Reinforcement Learning","date":"2023-10-03","arxiv_id":"2310.02360","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-design-principles-for-frequentist","slug":"bayesian-design-principles-for-frequentist","title":"Bayesian Design Principles for Frequentist Sequential Learning","date":"2023-10-01","arxiv_id":"2310.00806","repositories_listed":1,"syntology":null},{"url":"/paper/pre-training-with-synthetic-data-helps","slug":"pre-training-with-synthetic-data-helps","title":"Pre-training with Synthetic Data Helps Offline Reinforcement Learning","date":"2023-10-01","arxiv_id":"2310.00771","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pre-training-with-synthetic-data-helps#ran","syntology_url":"https://syntology.ai/paper/2310.00771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00771"}},"official":{"repos":["victor-wang-902/synthetic-pretrain-rl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cleanba-a-reproducible-and-efficient","slug":"cleanba-a-reproducible-and-efficient","title":"Cleanba: A Reproducible and Efficient Distributed Reinforcement Learning Platform","date":"2023-09-29","arxiv_id":"2310.00036","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/cleanba-a-reproducible-and-efficient#ran","syntology_url":"https://syntology.ai/paper/2310.00036","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00036"}},"official":{"repos":["vwxyzjn/cleanba"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/estimation-and-inference-in-distributional","slug":"estimation-and-inference-in-distributional","title":"Estimation and Inference in Distributional Reinforcement Learning","date":"2023-09-29","arxiv_id":"2309.17262","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/estimation-and-inference-in-distributional#ran","syntology_url":"https://syntology.ai/paper/2309.17262","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17262"}},"official":{"repos":["zhangliangyu32/estimationandinferencedistributionalrl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-terminate-in-object-navigation","slug":"learning-to-terminate-in-object-navigation","title":"Learning to Terminate in Object Navigation","date":"2023-09-28","arxiv_id":"2309.16164","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-streaming-video-temporal-action","slug":"end-to-end-streaming-video-temporal-action","title":"End-to-End Streaming Video Temporal Action Segmentation with Reinforce Learning","date":"2023-09-27","arxiv_id":"2309.15683","repositories_listed":1,"syntology":null},{"url":"/paper/towards-human-like-rl-taming-non-naturalistic","slug":"towards-human-like-rl-taming-non-naturalistic","title":"Towards Human-Like RL: Taming Non-Naturalistic Behavior in Deep RL via Adaptive Behavioral Costs in 3D Games","date":"2023-09-27","arxiv_id":"2309.15484","repositories_listed":1,"syntology":null},{"url":"/paper/maximum-diffusion-reinforcement-learning","slug":"maximum-diffusion-reinforcement-learning","title":"Maximum diffusion reinforcement learning","date":"2023-09-26","arxiv_id":"2309.15293","repositories_listed":1,"syntology":null},{"url":"/paper/tempo-adaptation-in-non-stationary-1","slug":"tempo-adaptation-in-non-stationary-1","title":"Tempo Adaptation in Non-stationary Reinforcement Learning","date":"2023-09-26","arxiv_id":"2309.14989","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/tempo-adaptation-in-non-stationary-1#ran","syntology_url":"https://syntology.ai/paper/2309.14989","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.14989"}},"official":{"repos":["hyunin-lee/TempoRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/designing-and-evaluating-an-online","slug":"designing-and-evaluating-an-online","title":"Designing and evaluating an online reinforcement learning agent for physical exercise recommendations in N-of-1 trials","date":"2023-09-25","arxiv_id":"2309.14156","repositories_listed":1,"syntology":null},{"url":"/paper/counterfactual-conservative-q-learning-for-1","slug":"counterfactual-conservative-q-learning-for-1","title":"Counterfactual Conservative Q Learning for Offline Multi-agent Reinforcement Learning","date":"2023-09-22","arxiv_id":"2309.12696","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":3,"n_ran_checked":4,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":9,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/counterfactual-conservative-q-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/2309.12696","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.12696"}},"official":{"repos":["thu-rllab/CFCQL"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/earnhft-efficient-hierarchical-reinforcement","slug":"earnhft-efficient-hierarchical-reinforcement","title":"EarnHFT: Efficient Hierarchical Reinforcement Learning for High Frequency Trading","date":"2023-09-22","arxiv_id":"2309.12891","repositories_listed":1,"syntology":null},{"url":"/paper/sequential-action-induced-invariant","slug":"sequential-action-induced-invariant","title":"Sequential Action-Induced Invariant Representation for Reinforcement Learning","date":"2023-09-22","arxiv_id":"2309.12628","repositories_listed":1,"syntology":null},{"url":"/paper/distributional-pareto-optimal-multi-objective","slug":"distributional-pareto-optimal-multi-objective","title":"Distributional Pareto-Optimal Multi-Objective Reinforcement Learning","date":"2023-09-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/representation-abstractions-as-incentives-for","slug":"representation-abstractions-as-incentives-for","title":"State Representations as Incentives for Reinforcement Learning Agents: A Sim2Real Analysis on Robotic Grasping","date":"2023-09-21","arxiv_id":"2309.11984","repositories_listed":1,"syntology":null},{"url":"/paper/text2reward-automated-dense-reward-function","slug":"text2reward-automated-dense-reward-function","title":"Text2Reward: Reward Shaping with Language Models for Reinforcement Learning","date":"2023-09-20","arxiv_id":"2309.11489","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/text2reward-automated-dense-reward-function#ran","syntology_url":"https://syntology.ai/paper/2309.11489","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.11489"}},"official":{"repos":["xlang-ai/text2reward"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/actively-learning-reinforcement-learning-a","slug":"actively-learning-reinforcement-learning-a","title":"Actively Learning Reinforcement Learning: A Stochastic Optimal Control Approach","date":"2023-09-18","arxiv_id":"2309.10831","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-initial-state-buffer-for","slug":"contrastive-initial-state-buffer-for","title":"Contrastive Initial State Buffer for Reinforcement Learning","date":"2023-09-18","arxiv_id":"2309.09752","repositories_listed":1,"syntology":null},{"url":"/paper/projected-task-specific-layers-for-multi-task","slug":"projected-task-specific-layers-for-multi-task","title":"Projected Task-Specific Layers for Multi-Task Reinforcement Learning","date":"2023-09-15","arxiv_id":"2309.08776","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-reinforcement-learning-for-jumping","slug":"efficient-reinforcement-learning-for-jumping","title":"Efficient Reinforcement Learning for Jumping Monopods","date":"2023-09-13","arxiv_id":"2309.07038","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforcement-learning-approach-for-robotic","slug":"a-reinforcement-learning-approach-for-robotic","title":"A Reinforcement Learning Approach for Robotic Unloading from Visual Observations","date":"2023-09-12","arxiv_id":"2309.06621","repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-with-latent-diffusion-in-offline","slug":"reasoning-with-latent-diffusion-in-offline","title":"Reasoning with Latent Diffusion in Offline Reinforcement Learning","date":"2023-09-12","arxiv_id":"2309.06599","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/reasoning-with-latent-diffusion-in-offline#ran","syntology_url":"https://syntology.ai/paper/2309.06599","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.06599"}},"official":{"repos":["ldcq/ldcq"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/subwords-as-skills-tokenization-for-sparse","slug":"subwords-as-skills-tokenization-for-sparse","title":"Subwords as Skills: Tokenization for Sparse-Reward Reinforcement Learning","date":"2023-09-08","arxiv_id":"2309.04459","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/subwords-as-skills-tokenization-for-sparse#ran","syntology_url":"https://syntology.ai/paper/2309.04459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.04459"}},"official":{"repos":["dyunis/subwords_as_skills"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-of-generalizable-and-interpretable","slug":"learning-of-generalizable-and-interpretable","title":"Learning of Generalizable and Interpretable Knowledge in Grid-Based Reinforcement Learning Environments","date":"2023-09-07","arxiv_id":"2309.03651","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-from-hierarchical","slug":"deep-reinforcement-learning-from-hierarchical","title":"Deep Reinforcement Learning from Hierarchical Preference Design","date":"2023-09-06","arxiv_id":"2309.02632","repositories_listed":1,"syntology":null},{"url":"/paper/orl-auditor-dataset-auditing-in-offline-deep","slug":"orl-auditor-dataset-auditing-in-offline-deep","title":"ORL-AUDITOR: Dataset Auditing in Offline Deep Reinforcement Learning","date":"2023-09-06","arxiv_id":"2309.03081","repositories_listed":1,"syntology":null},{"url":"/paper/rlsync-offline-online-reinforcement-learning","slug":"rlsync-offline-online-reinforcement-learning","title":"RLSynC: Offline-Online Reinforcement Learning for Synthon Completion","date":"2023-09-06","arxiv_id":"2309.02671","repositories_listed":1,"syntology":null},{"url":"/paper/autonomous-soft-tissue-retraction-using","slug":"autonomous-soft-tissue-retraction-using","title":"Autonomous Soft Tissue Retraction Using Demonstration-Guided Reinforcement Learning","date":"2023-09-02","arxiv_id":"2309.00837","repositories_listed":1,"syntology":null},{"url":"/paper/context-aware-composition-of-agent-policies","slug":"context-aware-composition-of-agent-policies","title":"Context-Aware Composition of Agent Policies by Markov Decision Process Entity Embeddings and Agent Ensembles","date":"2023-08-28","arxiv_id":"2308.14521","repositories_listed":1,"syntology":null},{"url":"/paper/edge-generation-scheduling-for-dag-tasks","slug":"edge-generation-scheduling-for-dag-tasks","title":"Edge Generation Scheduling for DAG Tasks Using Deep Reinforcement Learning","date":"2023-08-28","arxiv_id":"2308.14647","repositories_listed":1,"syntology":null},{"url":"/paper/learning-visual-tracking-and-reaching-with","slug":"learning-visual-tracking-and-reaching-with","title":"Learning Visual Tracking and Reaching with Deep Reinforcement Learning on a UR10e Robotic Arm","date":"2023-08-28","arxiv_id":"2308.14652","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-sampling-on","slug":"reinforcement-learning-for-sampling-on","title":"Reinforcement Learning for Sampling on Temporal Medical Imaging Sequences","date":"2023-08-28","arxiv_id":"2308.14946","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/reinforcement-learning-for-sampling-on#ran","syntology_url":"https://syntology.ai/paper/2308.14946","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.14946"}},"official":{"repos":["zhishenhuang/rlsamp"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/simple-modification-of-the-upper-confidence","slug":"simple-modification-of-the-upper-confidence","title":"Simple Modification of the Upper Confidence Bound Algorithm by Generalized Weighted Averages","date":"2023-08-28","arxiv_id":"2308.14350","repositories_listed":1,"syntology":null},{"url":"/paper/traffic-light-control-with-reinforcement","slug":"traffic-light-control-with-reinforcement","title":"Traffic Light Control with Reinforcement Learning","date":"2023-08-28","arxiv_id":"2308.14295","repositories_listed":1,"syntology":null},{"url":"/paper/learning-collaborative-information","slug":"learning-collaborative-information","title":"Collaborative Information Dissemination with Graph-based Multi-Agent Reinforcement Learning","date":"2023-08-25","arxiv_id":"2308.16198","repositories_listed":1,"syntology":null},{"url":"/paper/molopt-autonomous-molecular-geometry","slug":"molopt-autonomous-molecular-geometry","title":"MolOpt: Autonomous Molecular Geometry Optimization using Multi-Agent Reinforcement Learning","date":"2023-08-24","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/deploying-deep-reinforcement-learning-systems","slug":"deploying-deep-reinforcement-learning-systems","title":"Deploying Deep Reinforcement Learning Systems: A Taxonomy of Challenges","date":"2023-08-23","arxiv_id":"2308.12438","repositories_listed":1,"syntology":null},{"url":"/paper/diverse-policies-converge-in-reward-free","slug":"diverse-policies-converge-in-reward-free","title":"Diverse Policies Converge in Reward-free Markov Decision Processe","date":"2023-08-23","arxiv_id":"2308.11924","repositories_listed":1,"syntology":null},{"url":"/paper/language-reward-modulation-for-pretraining","slug":"language-reward-modulation-for-pretraining","title":"Language Reward Modulation for Pretraining Reinforcement Learning","date":"2023-08-23","arxiv_id":"2308.12270","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-reward-modulation-for-pretraining#ran","syntology_url":"https://syntology.ai/paper/2308.12270","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12270"}},"official":{"repos":["ademiadeniji/lamp"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fox-formation-aware-exploration-in-multi","slug":"fox-formation-aware-exploration-in-multi","title":"FoX: Formation-aware exploration in multi-agent reinforcement learning","date":"2023-08-22","arxiv_id":"2308.11272","repositories_listed":1,"syntology":null},{"url":"/paper/lagr-seq-language-guided-reinforcement","slug":"lagr-seq-language-guided-reinforcement","title":"LaGR-SEQ: Language-Guided Reinforcement Learning with Sample-Efficient Querying","date":"2023-08-21","arxiv_id":"2308.13542","repositories_listed":1,"syntology":null},{"url":"/paper/dpmac-differentially-private-communication","slug":"dpmac-differentially-private-communication","title":"DPMAC: Differentially Private Communication for Cooperative Multi-Agent Reinforcement Learning","date":"2023-08-19","arxiv_id":"2308.09902","repositories_listed":1,"syntology":null},{"url":"/paper/discrete-prompt-compression-with","slug":"discrete-prompt-compression-with","title":"Discrete Prompt Compression with Reinforcement Learning","date":"2023-08-17","arxiv_id":"2308.08758","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/discrete-prompt-compression-with#ran","syntology_url":"https://syntology.ai/paper/2308.08758","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.08758"}},"official":{"repos":["nenomigami/promptcompressor"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-identify-critical-states-for","slug":"learning-to-identify-critical-states-for","title":"Learning to Identify Critical States for Reinforcement Learning from Videos","date":"2023-08-15","arxiv_id":"2308.07795","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-identify-critical-states-for#ran","syntology_url":"https://syntology.ai/paper/2308.07795","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.07795"}},"official":{"repos":["ai-initiative-kaust/videorlcs"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-in-financial-markets-a","slug":"reinforcement-learning-in-financial-markets-a","title":"Reinforcement Learning in Financial Markets: A Study on Dynamic Model Weight Assignment","date":"2023-08-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/heterogeneous-multi-agent-reinforcement-1","slug":"heterogeneous-multi-agent-reinforcement-1","title":"Heterogeneous Multi-Agent Reinforcement Learning via Mirror Descent Policy Optimization","date":"2023-08-13","arxiv_id":"2308.06741","repositories_listed":1,"syntology":null},{"url":"/paper/value-distributional-model-based","slug":"value-distributional-model-based","title":"Value-Distributional Model-Based Reinforcement Learning","date":"2023-08-12","arxiv_id":"2308.06590","repositories_listed":1,"syntology":null},{"url":"/paper/rlsac-reinforcement-learning-enhanced-sample","slug":"rlsac-reinforcement-learning-enhanced-sample","title":"RLSAC: Reinforcement Learning enhanced Sample Consensus for End-to-End Robust Estimation","date":"2023-08-10","arxiv_id":"2308.05318","repositories_listed":1,"syntology":null},{"url":"/paper/barlowrl-barlow-twins-for-data-efficient","slug":"barlowrl-barlow-twins-for-data-efficient","title":"BarlowRL: Barlow Twins for Data-Efficient Reinforcement Learning","date":"2023-08-08","arxiv_id":"2308.04263","repositories_listed":1,"syntology":null},{"url":"/paper/alphastar-unplugged-large-scale-offline","slug":"alphastar-unplugged-large-scale-offline","title":"AlphaStar Unplugged: Large-Scale Offline Reinforcement Learning","date":"2023-08-07","arxiv_id":"2308.03526","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/alphastar-unplugged-large-scale-offline#ran","syntology_url":"https://syntology.ai/paper/2308.03526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.03526"}},"official":null}},{"url":"/paper/reinforcement-learning-for-financial-index","slug":"reinforcement-learning-for-financial-index","title":"Reinforcement Learning for Financial Index Tracking","date":"2023-08-05","arxiv_id":"2308.02820","repositories_listed":1,"syntology":null},{"url":"/paper/job-shop-scheduling-via-deep-reinforcement","slug":"job-shop-scheduling-via-deep-reinforcement","title":"Job Shop Scheduling via Deep Reinforcement Learning: a Sequence to Sequence approach","date":"2023-08-03","arxiv_id":"2308.01797","repositories_listed":1,"syntology":null},{"url":"/paper/smarla-a-safety-monitoring-approach-for-deep","slug":"smarla-a-safety-monitoring-approach-for-deep","title":"SMARLA: A Safety Monitoring Approach for Deep Reinforcement Learning Agents","date":"2023-08-03","arxiv_id":"2308.02594","repositories_listed":1,"syntology":null},{"url":"/paper/bierl-a-meta-evolutionary-reinforcement","slug":"bierl-a-meta-evolutionary-reinforcement","title":"BiERL: A Meta Evolutionary Reinforcement Learning Framework via Bilevel Optimization","date":"2023-08-01","arxiv_id":"2308.01207","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-non","slug":"reinforcement-learning-based-non","title":"Reinforcement Learning-based Non-Autoregressive Solver for Traveling Salesman Problems","date":"2023-08-01","arxiv_id":"2308.00560","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-based-non#ran","syntology_url":"https://syntology.ai/paper/2308.00560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.00560"}},"official":{"repos":["xybfight/nar4tsp"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-multi-agent-reinforcement-learning-3","slug":"robust-multi-agent-reinforcement-learning-3","title":"Robust Multi-Agent Reinforcement Learning with State Uncertainty","date":"2023-07-30","arxiv_id":"2307.16212","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/robust-multi-agent-reinforcement-learning-3#ran","syntology_url":"https://syntology.ai/paper/2307.16212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.16212"}},"official":{"repos":["sihongho/robust_marl_with_state_uncertainty"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/variance-control-for-distributional","slug":"variance-control-for-distributional","title":"Variance Control for Distributional Reinforcement Learning","date":"2023-07-30","arxiv_id":"2307.16152","repositories_listed":1,"syntology":null},{"url":"/paper/curiosity-driven-reinforcement-learning-based","slug":"curiosity-driven-reinforcement-learning-based","title":"Curiosity-Driven Reinforcement Learning based Low-Level Flight Control","date":"2023-07-28","arxiv_id":"2307.15724","repositories_listed":1,"syntology":null},{"url":"/paper/approximate-model-based-shielding-for-safe","slug":"approximate-model-based-shielding-for-safe","title":"Approximate Model-Based Shielding for Safe Reinforcement Learning","date":"2023-07-27","arxiv_id":"2308.00707","repositories_listed":1,"syntology":null},{"url":"/paper/flare-fingerprinting-deep-reinforcement","slug":"flare-fingerprinting-deep-reinforcement","title":"FLARE: Fingerprinting Deep Reinforcement Learning Agents using Universal Adversarial Masks","date":"2023-07-27","arxiv_id":"2307.14751","repositories_listed":1,"syntology":null},{"url":"/paper/mode-constrained-model-based-reinforcement","slug":"mode-constrained-model-based-reinforcement","title":"Mode-constrained Model-based Reinforcement Learning via Gaussian Processes","date":"2023-07-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/submodular-reinforcement-learning","slug":"submodular-reinforcement-learning","title":"Submodular Reinforcement Learning","date":"2023-07-25","arxiv_id":"2307.13372","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/submodular-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2307.13372","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.13372"}},"official":{"repos":["manish-pra/non-additive-rl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/balancing-exploration-and-exploitation-in","slug":"balancing-exploration-and-exploitation-in","title":"Balancing Exploration and Exploitation in Hierarchical Reinforcement Learning via Latent Landmark Graphs","date":"2023-07-22","arxiv_id":"2307.12063","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/balancing-exploration-and-exploitation-in#ran","syntology_url":"https://syntology.ai/paper/2307.12063","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12063"}},"official":{"repos":["papercode2022/hill"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/emergence-of-adaptive-circadian-rhythms-in","slug":"emergence-of-adaptive-circadian-rhythms-in","title":"Emergence of Adaptive Circadian Rhythms in Deep Reinforcement Learning","date":"2023-07-22","arxiv_id":"2307.12143","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emergence-of-adaptive-circadian-rhythms-in#ran","syntology_url":"https://syntology.ai/paper/2307.12143","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12143"}},"official":{"repos":["aqeel13932/mn_project"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hindsight-dice-stable-credit-assignment-for","slug":"hindsight-dice-stable-credit-assignment-for","title":"Hindsight-DICE: Stable Credit Assignment for Deep Reinforcement Learning","date":"2023-07-21","arxiv_id":"2307.11897","repositories_listed":1,"syntology":null},{"url":"/paper/joingym-an-efficient-query-optimization","slug":"joingym-an-efficient-query-optimization","title":"JoinGym: An Efficient Query Optimization Environment for Reinforcement Learning","date":"2023-07-21","arxiv_id":"2307.11704","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-offline-reinforcement-learning-2","slug":"model-based-offline-reinforcement-learning-2","title":"Model-based Offline Reinforcement Learning with Count-based Conservatism","date":"2023-07-21","arxiv_id":"2307.11352","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":1,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/model-based-offline-reinforcement-learning-2#ran","syntology_url":"https://syntology.ai/paper/2307.11352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11352"}},"official":{"repos":["oh-lab/count-morl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/breadcrumbs-to-the-goal-goal-conditioned","slug":"breadcrumbs-to-the-goal-goal-conditioned","title":"Breadcrumbs to the Goal: Goal-Conditioned Exploration from Human-in-the-Loop Feedback","date":"2023-07-20","arxiv_id":"2307.11049","repositories_listed":1,"syntology":null},{"url":"/paper/longitudinal-data-and-a-semantic-similarity","slug":"longitudinal-data-and-a-semantic-similarity","title":"Longitudinal Data and a Semantic Similarity Reward for Chest X-Ray Report Generation","date":"2023-07-19","arxiv_id":"2307.09758","repositories_listed":1,"syntology":null},{"url":"/paper/pytag-challenges-and-opportunities-for","slug":"pytag-challenges-and-opportunities-for","title":"PyTAG: Challenges and Opportunities for Reinforcement Learning in Tabletop Games","date":"2023-07-19","arxiv_id":"2307.09905","repositories_listed":1,"syntology":null},{"url":"/paper/natural-actor-critic-for-robust-reinforcement","slug":"natural-actor-critic-for-robust-reinforcement","title":"Natural Actor-Critic for Robust Reinforcement Learning with Function Approximation","date":"2023-07-17","arxiv_id":"2307.08875","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":8,"n_ran_checked":9,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":14,"phrase":"11 ran (of which 8 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/natural-actor-critic-for-robust-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2307.08875","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08875"}},"official":{"repos":["tliu1997/rnac"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":8,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}}],"record_sha256":"2b26e7e2ed5d49cbf38112c835f36948e9b5d3ea6157025523018ed7df80026c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}