{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/19","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":19,"pages_in_order":132,"rows_per_page":100,"rows":[1801,1900],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/18","next":"/task/reinforcement-learning/papers/20","papers":[{"url":"/paper/rl-vigen-a-reinforcement-learning-benchmark-1","slug":"rl-vigen-a-reinforcement-learning-benchmark-1","title":"RL-ViGen: A Reinforcement Learning Benchmark for Visual Generalization","date":"2023-07-15","arxiv_id":"2307.10224","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rl-vigen-a-reinforcement-learning-benchmark-1#ran","syntology_url":"https://syntology.ai/paper/2307.10224","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.10224"}},"official":{"repos":["gemcollector/rl-vigen"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-photonic-component","slug":"reinforcement-learning-for-photonic-component","title":"Reinforcement Learning for Photonic Component Design","date":"2023-07-14","arxiv_id":"2307.11075","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-frontier-based","slug":"reinforcement-learning-with-frontier-based","title":"Reinforcement Learning with Frontier-Based Exploration via Autonomous Environment","date":"2023-07-14","arxiv_id":"2307.07296","repositories_listed":1,"syntology":null},{"url":"/paper/safe-dreamerv3-safe-reinforcement-learning","slug":"safe-dreamerv3-safe-reinforcement-learning","title":"SafeDreamer: Safe Reinforcement Learning with World Models","date":"2023-07-14","arxiv_id":"2307.07176","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":4,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/safe-dreamerv3-safe-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2307.07176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.07176"}},"official":{"repos":["pku-alignment/safedreamer"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/prescriptive-process-monitoring-under-1","slug":"prescriptive-process-monitoring-under-1","title":"Prescriptive Process Monitoring Under Resource Constraints: A Reinforcement Learning Approach","date":"2023-07-13","arxiv_id":"2307.06564","repositories_listed":1,"syntology":null},{"url":"/paper/robotic-manipulation-datasets-for-offline","slug":"robotic-manipulation-datasets-for-offline","title":"Robotic Manipulation Datasets for Offline Compositional Reinforcement Learning","date":"2023-07-13","arxiv_id":"2307.07091","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robotic-manipulation-datasets-for-offline#ran","syntology_url":"https://syntology.ai/paper/2307.07091","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.07091"}},"official":{"repos":["lifelong-ml/offline-compositional-rl-datasets"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-reinforcement-learning-as-wasserstein","slug":"safe-reinforcement-learning-as-wasserstein","title":"Probabilistic Constrained Reinforcement Learning with Formal Interpretability","date":"2023-07-13","arxiv_id":"2307.07084","repositories_listed":1,"syntology":null},{"url":"/paper/dsse-a-drone-swarm-search-environment","slug":"dsse-a-drone-swarm-search-environment","title":"DSSE: a drone swarm search environment","date":"2023-07-12","arxiv_id":"2307.06240","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-feedback-efficiency-of-interactive","slug":"boosting-feedback-efficiency-of-interactive","title":"Boosting Feedback Efficiency of Interactive Reinforcement Learning by Adaptive Learning from Scores","date":"2023-07-11","arxiv_id":"2307.05405","repositories_listed":1,"syntology":null},{"url":"/paper/intrinsically-motivated-graph-exploration","slug":"intrinsically-motivated-graph-exploration","title":"Intrinsically motivated graph exploration using network theories of human curiosity","date":"2023-07-11","arxiv_id":"2307.04962","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-non-cumulative","slug":"reinforcement-learning-with-non-cumulative","title":"Reinforcement Learning with Non-Cumulative Objective","date":"2023-07-11","arxiv_id":"2307.04957","repositories_listed":1,"syntology":null},{"url":"/paper/loss-dynamics-of-temporal-difference","slug":"loss-dynamics-of-temporal-difference","title":"Loss Dynamics of Temporal Difference Reinforcement Learning","date":"2023-07-10","arxiv_id":"2307.04841","repositories_listed":1,"syntology":null},{"url":"/paper/probabilistic-counterexample-guidance-for","slug":"probabilistic-counterexample-guidance-for","title":"Probabilistic Counterexample Guidance for Safer Reinforcement Learning (Extended Version)","date":"2023-07-10","arxiv_id":"2307.04927","repositories_listed":1,"syntology":null},{"url":"/paper/rltf-reinforcement-learning-from-unit-test","slug":"rltf-reinforcement-learning-from-unit-test","title":"RLTF: Reinforcement Learning from Unit Test Feedback","date":"2023-07-10","arxiv_id":"2307.04349","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/rltf-reinforcement-learning-from-unit-test#ran","syntology_url":"https://syntology.ai/paper/2307.04349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.04349"}},"official":{"repos":["zyq-scut/rltf"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/discovering-hierarchical-achievements-in-1","slug":"discovering-hierarchical-achievements-in-1","title":"Discovering Hierarchical Achievements in Reinforcement Learning via Contrastive Learning","date":"2023-07-07","arxiv_id":"2307.03486","repositories_listed":1,"syntology":null},{"url":"/paper/containergym-a-real-world-reinforcement","slug":"containergym-a-real-world-reinforcement","title":"ContainerGym: A Real-World Reinforcement Learning Benchmark for Resource Allocation","date":"2023-07-06","arxiv_id":"2307.02991","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-observation-policies-in-observation","slug":"dynamic-observation-policies-in-observation","title":"Dynamic Observation Policies in Observation Cost-Sensitive Reinforcement Learning","date":"2023-07-05","arxiv_id":"2307.02620","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/dynamic-observation-policies-in-observation#ran","syntology_url":"https://syntology.ai/paper/2307.02620","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.02620"}},"official":{"repos":["cbellinger27/learning-when-to-observe-in-rl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/learning-symbolic-rules-over-abstract-meaning","slug":"learning-symbolic-rules-over-abstract-meaning","title":"Learning Symbolic Rules over Abstract Meaning Representations for Textual Reinforcement Learning","date":"2023-07-05","arxiv_id":"2307.02689","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-symbolic-rules-over-abstract-meaning#ran","syntology_url":"https://syntology.ai/paper/2307.02689","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.02689"}},"official":{"repos":["ibm/loa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-objective-deep-reinforcement-learning-1","slug":"multi-objective-deep-reinforcement-learning-1","title":"Multi-objective Deep Reinforcement Learning for Mobile Edge Computing","date":"2023-07-05","arxiv_id":"2307.14346","repositories_listed":1,"syntology":null},{"url":"/paper/deep-attention-q-network-for-personalized","slug":"deep-attention-q-network-for-personalized","title":"Deep Attention Q-Network for Personalized Treatment Recommendation","date":"2023-07-04","arxiv_id":"2307.01519","repositories_listed":1,"syntology":null},{"url":"/paper/distributional-model-equivalence-for-risk","slug":"distributional-model-equivalence-for-risk","title":"Distributional Model Equivalence for Risk-Sensitive Reinforcement Learning","date":"2023-07-04","arxiv_id":"2307.01708","repositories_listed":1,"syntology":null},{"url":"/paper/environmental-effects-on-emergent-strategy-in","slug":"environmental-effects-on-emergent-strategy-in","title":"Environmental effects on emergent strategy in micro-scale multi-agent reinforcement learning","date":"2023-07-03","arxiv_id":"2307.00994","repositories_listed":1,"syntology":null},{"url":"/paper/safe-model-based-multi-agent-mean-field","slug":"safe-model-based-multi-agent-mean-field","title":"Safe Model-Based Multi-Agent Mean-Field Reinforcement Learning","date":"2023-06-29","arxiv_id":"2306.17052","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-truss-design-with-reinforcement","slug":"automatic-truss-design-with-reinforcement","title":"Automatic Truss Design with Reinforcement Learning","date":"2023-06-27","arxiv_id":"2306.15182","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-ood-state-actions-supported-cross","slug":"beyond-ood-state-actions-supported-cross","title":"Beyond OOD State Actions: Supported Cross-Domain Offline Reinforcement Learning","date":"2023-06-22","arxiv_id":"2306.12755","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-ood-state-actions-supported-cross#ran","syntology_url":"https://syntology.ai/paper/2306.12755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.12755"}},"official":{"repos":["thuml/SPOT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/harnessing-mixed-offline-reinforcement","slug":"harnessing-mixed-offline-reinforcement","title":"Harnessing Mixed Offline Reinforcement Learning Datasets via Trajectory Weighting","date":"2023-06-22","arxiv_id":"2306.13085","repositories_listed":1,"syntology":null},{"url":"/paper/state-wise-constrained-policy-optimization","slug":"state-wise-constrained-policy-optimization","title":"State-wise Constrained Policy Optimization","date":"2023-06-21","arxiv_id":"2306.12594","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-ordered-information-extraction-with","slug":"adaptive-ordered-information-extraction-with","title":"Adaptive Ordered Information Extraction with Deep Reinforcement Learning","date":"2023-06-19","arxiv_id":"2306.10787","repositories_listed":1,"syntology":null},{"url":"/paper/cammarl-conformal-action-modeling-in-multi","slug":"cammarl-conformal-action-modeling-in-multi","title":"CAMMARL: Conformal Action Modeling in Multi Agent Reinforcement Learning","date":"2023-06-19","arxiv_id":"2306.11128","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/cammarl-conformal-action-modeling-in-multi#ran","syntology_url":"https://syntology.ai/paper/2306.11128","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.11128"}},"official":{"repos":["nikunj-gupta/conformal-agent-modelling"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/deep-reinforcement-learning-with-multitask","slug":"deep-reinforcement-learning-with-multitask","title":"Deep Reinforcement Learning with Task-Adaptive Retrieval via Hypernetwork","date":"2023-06-19","arxiv_id":"2306.10698","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-quantum-variational-state","slug":"enhancing-quantum-variational-state","title":"Enhancing variational quantum state diagonalization using reinforcement learning techniques","date":"2023-06-19","arxiv_id":"2306.11086","repositories_listed":1,"syntology":null},{"url":"/paper/maximum-entropy-heterogeneous-agent-mirror","slug":"maximum-entropy-heterogeneous-agent-mirror","title":"Maximum Entropy Heterogeneous-Agent Reinforcement Learning","date":"2023-06-19","arxiv_id":"2306.10715","repositories_listed":1,"syntology":null},{"url":"/paper/empowering-nlg-offline-reinforcement-learning-1","slug":"empowering-nlg-offline-reinforcement-learning-1","title":"Empowering NLG: Offline Reinforcement Learning for Informal Summarization in Online Domains","date":"2023-06-17","arxiv_id":"2306.17174","repositories_listed":1,"syntology":null},{"url":"/paper/genes-in-intelligent-agents","slug":"genes-in-intelligent-agents","title":"Genes in Intelligent Agents","date":"2023-06-17","arxiv_id":"2306.10225","repositories_listed":1,"syntology":null},{"url":"/paper/creating-multi-level-skill-hierarchies-in","slug":"creating-multi-level-skill-hierarchies-in","title":"Creating Multi-Level Skill Hierarchies in Reinforcement Learning","date":"2023-06-16","arxiv_id":"2306.09980","repositories_listed":1,"syntology":null},{"url":"/paper/semi-offline-reinforcement-learning-for","slug":"semi-offline-reinforcement-learning-for","title":"Semi-Offline Reinforcement Learning for Optimized Text Generation","date":"2023-06-16","arxiv_id":"2306.09712","repositories_listed":1,"syntology":null},{"url":"/paper/a-framework-for-learning-from-demonstration","slug":"a-framework-for-learning-from-demonstration","title":"A Framework for Learning from Demonstration with Minimal Human Effort","date":"2023-06-15","arxiv_id":"2306.09211","repositories_listed":1,"syntology":null},{"url":"/paper/generalizable-resource-scaling-of-5g-slices","slug":"generalizable-resource-scaling-of-5g-slices","title":"Generalizable Resource Scaling of 5G Slices using Constrained Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.09290","repositories_listed":1,"syntology":null},{"url":"/paper/joint-path-planning-and-power-allocation-of-a","slug":"joint-path-planning-and-power-allocation-of-a","title":"Joint Path planning and Power Allocation of a Cellular-Connected UAV using Apprenticeship Learning via Deep Inverse Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.10071","repositories_listed":1,"syntology":null},{"url":"/paper/recurrent-memory-decision-transformer","slug":"recurrent-memory-decision-transformer","title":"Recurrent Action Transformer with Memory","date":"2023-06-15","arxiv_id":"2306.09459","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recurrent-memory-decision-transformer#ran","syntology_url":"https://syntology.ai/paper/2306.09459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09459"}},"official":{"repos":["airi-institute/rate"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/semantic-helm-a-human-readable-memory-for-1","slug":"semantic-helm-a-human-readable-memory-for-1","title":"Semantic HELM: A Human-Readable Memory for Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.09312","repositories_listed":1,"syntology":null},{"url":"/paper/simplified-temporal-consistency-reinforcement","slug":"simplified-temporal-consistency-reinforcement","title":"Simplified Temporal Consistency Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.09466","repositories_listed":1,"syntology":null},{"url":"/paper/curricular-subgoals-for-inverse-reinforcement","slug":"curricular-subgoals-for-inverse-reinforcement","title":"Curricular Subgoals for Inverse Reinforcement Learning","date":"2023-06-14","arxiv_id":"2306.08232","repositories_listed":1,"syntology":null},{"url":"/paper/katakomba-tools-and-benchmarks-for-data","slug":"katakomba-tools-and-benchmarks-for-data","title":"Katakomba: Tools and Benchmarks for Data-Driven NetHack","date":"2023-06-14","arxiv_id":"2306.08772","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/katakomba-tools-and-benchmarks-for-data#ran","syntology_url":"https://syntology.ai/paper/2306.08772","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08772"}},"official":{"repos":["corl-team/katakomba"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mediated-multi-agent-reinforcement-learning","slug":"mediated-multi-agent-reinforcement-learning","title":"Mediated Multi-Agent Reinforcement Learning","date":"2023-06-14","arxiv_id":"2306.08419","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mediated-multi-agent-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2306.08419","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08419"}},"official":{"repos":["dimonenka/mediatedmarl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/ocatari-object-centric-atari-2600","slug":"ocatari-object-centric-atari-2600","title":"OCAtari: Object-Centric Atari 2600 Reinforcement Learning Environments","date":"2023-06-14","arxiv_id":"2306.08649","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ocatari-object-centric-atari-2600#ran","syntology_url":"https://syntology.ai/paper/2306.08649","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08649"}},"official":{"repos":["k4ntz/oc_atari"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-versatile-multi-agent-reinforcement","slug":"a-versatile-multi-agent-reinforcement","title":"A Versatile Multi-Agent Reinforcement Learning Benchmark for Inventory Management","date":"2023-06-13","arxiv_id":"2306.07542","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-versatile-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2306.07542","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.07542"}},"official":{"repos":["victoryxl/replenishmentenv"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generalizable-wireless-navigation-through","slug":"generalizable-wireless-navigation-through","title":"Digital Twin-Enhanced Wireless Indoor Navigation: Achieving Efficient Environment Sensing with Zero-Shot Reinforcement Learning","date":"2023-06-11","arxiv_id":"2306.06766","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalizable-wireless-navigation-through#ran","syntology_url":"https://syntology.ai/paper/2306.06766","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.06766"}},"official":{"repos":["panshark/pirl-win"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-efficacy-of-3d-point-cloud","slug":"on-the-efficacy-of-3d-point-cloud","title":"On the Efficacy of 3D Point Cloud Reinforcement Learning","date":"2023-06-11","arxiv_id":"2306.06799","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/on-the-efficacy-of-3d-point-cloud#ran","syntology_url":"https://syntology.ai/paper/2306.06799","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.06799"}},"official":{"repos":["lz1oceani/pointcloud_rl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/decision-stacks-flexible-reinforcement","slug":"decision-stacks-flexible-reinforcement","title":"Decision Stacks: Flexible Reinforcement Learning via Modular Generative Models","date":"2023-06-09","arxiv_id":"2306.06253","repositories_listed":1,"syntology":null},{"url":"/paper/explaining-reinforcement-learning-with","slug":"explaining-reinforcement-learning-with","title":"Explaining Reinforcement Learning with Shapley Values","date":"2023-06-09","arxiv_id":"2306.05810","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/explaining-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2306.05810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.05810"}},"official":{"repos":["bath-reinforcement-learning-lab/sverl_icml_2023"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-are-semi-parametric-1","slug":"large-language-models-are-semi-parametric-1","title":"Large Language Models Are Semi-Parametric Reinforcement Learning Agents","date":"2023-06-09","arxiv_id":"2306.07929","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-are-semi-parametric-1#ran","syntology_url":"https://syntology.ai/paper/2306.07929","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.07929"}},"official":{"repos":["opendfm/rememberer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/treedqn-learning-to-minimize-branch-and-bound","slug":"treedqn-learning-to-minimize-branch-and-bound","title":"TreeDQN: Learning to minimize Branch-and-Bound tree","date":"2023-06-09","arxiv_id":"2306.05905","repositories_listed":1,"syntology":null},{"url":"/paper/learned-spatial-data-partitioning","slug":"learned-spatial-data-partitioning","title":"Learned spatial data partitioning","date":"2023-06-08","arxiv_id":"2306.04846","repositories_listed":1,"syntology":null},{"url":"/paper/agent-performing-autonomous-stock-trading","slug":"agent-performing-autonomous-stock-trading","title":"Agent Performing Autonomous Stock Trading under Good and Bad Situations","date":"2023-06-06","arxiv_id":"2306.03985","repositories_listed":1,"syntology":null},{"url":"/paper/backproptools-a-fast-portable-deep","slug":"backproptools-a-fast-portable-deep","title":"RLtools: A Fast, Portable Deep Reinforcement Learning Library for Continuous Control","date":"2023-06-06","arxiv_id":"2306.03530","repositories_listed":1,"syntology":null},{"url":"/paper/mildly-constrained-evaluation-policy-for","slug":"mildly-constrained-evaluation-policy-for","title":"Mildly Constrained Evaluation Policy for Offline Reinforcement Learning","date":"2023-06-06","arxiv_id":"2306.03680","repositories_listed":1,"syntology":null},{"url":"/paper/vid2act-activate-offline-videos-for-visual-rl","slug":"vid2act-activate-offline-videos-for-visual-rl","title":"Model-Based Reinforcement Learning with Multi-Task Offline Pretraining","date":"2023-06-06","arxiv_id":"2306.03360","repositories_listed":1,"syntology":null},{"url":"/paper/flipping-coins-to-estimate-pseudocounts-for","slug":"flipping-coins-to-estimate-pseudocounts-for","title":"Flipping Coins to Estimate Pseudocounts for Exploration in Reinforcement Learning","date":"2023-06-05","arxiv_id":"2306.03186","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/flipping-coins-to-estimate-pseudocounts-for#ran","syntology_url":"https://syntology.ai/paper/2306.03186","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03186"}},"official":{"repos":["samlobel/cfn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tackling-non-stationarity-in-reinforcement","slug":"tackling-non-stationarity-in-reinforcement","title":"Tackling Non-Stationarity in Reinforcement Learning via Causal-Origin Representation","date":"2023-06-05","arxiv_id":"2306.02747","repositories_listed":1,"syntology":null},{"url":"/paper/ma2cl-masked-attentive-contrastive-learning","slug":"ma2cl-masked-attentive-contrastive-learning","title":"MA2CL:Masked Attentive Contrastive Learning for Multi-Agent Reinforcement Learning","date":"2023-06-03","arxiv_id":"2306.02006","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":4,"n_ran_checked":5,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ma2cl-masked-attentive-contrastive-learning#ran","syntology_url":"https://syntology.ai/paper/2306.02006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.02006"}},"official":{"repos":["ustchlsong/ma2cl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-framework-for-1","slug":"deep-reinforcement-learning-framework-for-1","title":"Deep Reinforcement Learning Framework for Thoracic Diseases Classification via Prior Knowledge Guidance","date":"2023-06-02","arxiv_id":"2306.01232","repositories_listed":1,"syntology":null},{"url":"/paper/hyperparameters-in-reinforcement-learning-and","slug":"hyperparameters-in-reinforcement-learning-and","title":"Hyperparameters in Reinforcement Learning and How To Tune Them","date":"2023-06-02","arxiv_id":"2306.01324","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hyperparameters-in-reinforcement-learning-and#ran","syntology_url":"https://syntology.ai/paper/2306.01324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01324"}},"official":{"repos":["facebookresearch/how-to-autorl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tackling-unbounded-state-spaces-in-continuing","slug":"tackling-unbounded-state-spaces-in-continuing","title":"Learning to Stabilize Online Reinforcement Learning in Unbounded State Spaces","date":"2023-06-02","arxiv_id":"2306.01896","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":1,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":2,"n_no_contract":1,"n_pointer_only":6,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 2 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tackling-unbounded-state-spaces-in-continuing#ran","syntology_url":"https://syntology.ai/paper/2306.01896","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01896"}},"official":{"repos":["badger-rl/stop"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/identifiability-and-generalizability-in","slug":"identifiability-and-generalizability-in","title":"Identifiability and Generalizability in Constrained Inverse Reinforcement Learning","date":"2023-06-01","arxiv_id":"2306.00629","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/identifiability-and-generalizability-in#ran","syntology_url":"https://syntology.ai/paper/2306.00629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00629"}},"official":{"repos":["andrschl/cirl"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-and-benchmarking-offline","slug":"improving-and-benchmarking-offline","title":"Improving and Benchmarking Offline Reinforcement Learning Algorithms","date":"2023-06-01","arxiv_id":"2306.00972","repositories_listed":1,"syntology":null},{"url":"/paper/normalization-enhances-generalization-in","slug":"normalization-enhances-generalization-in","title":"Normalization Enhances Generalization in Visual Reinforcement Learning","date":"2023-06-01","arxiv_id":"2306.00656","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/normalization-enhances-generalization-in#ran","syntology_url":"https://syntology.ai/paper/2306.00656","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00656"}},"official":{"repos":["lilucse/Normalization-Enhances-Generalization-in-Visual-Reinforcement-Learning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-offline-reinforcement-learning-with-real","slug":"safe-offline-reinforcement-learning-with-real","title":"Safe Offline Reinforcement Learning with Real-Time Budget Constraints","date":"2023-06-01","arxiv_id":"2306.00603","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-diffusion-policies-for-offline-1","slug":"efficient-diffusion-policies-for-offline-1","title":"Efficient Diffusion Policies for Offline Reinforcement Learning","date":"2023-05-31","arxiv_id":"2305.20081","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-diffusion-policies-for-offline-1#ran","syntology_url":"https://syntology.ai/paper/2305.20081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.20081"}},"official":{"repos":["sail-sg/edp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/latent-exploration-for-reinforcement-learning","slug":"latent-exploration-for-reinforcement-learning","title":"Latent Exploration for Reinforcement Learning","date":"2023-05-31","arxiv_id":"2305.20065","repositories_listed":1,"syntology":null},{"url":"/paper/offline-meta-reinforcement-learning-with-in","slug":"offline-meta-reinforcement-learning-with-in","title":"Offline Meta Reinforcement Learning with In-Distribution Online Adaptation","date":"2023-05-31","arxiv_id":"2305.19529","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/offline-meta-reinforcement-learning-with-in#ran","syntology_url":"https://syntology.ai/paper/2305.19529","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.19529"}},"official":{"repos":["nagisazj/idaq_public"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rosarl-reward-only-safe-reinforcement","slug":"rosarl-reward-only-safe-reinforcement","title":"ROSARL: Reward-Only Safe Reinforcement Learning","date":"2023-05-31","arxiv_id":"2306.00035","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rosarl-reward-only-safe-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2306.00035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00035"}},"official":{"repos":["geraudnt/rosarl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-reinforcement-learning-objectives-for","slug":"robust-reinforcement-learning-objectives-for","title":"Robust Reinforcement Learning Objectives for Sequential Recommender Systems","date":"2023-05-30","arxiv_id":"2305.18820","repositories_listed":1,"syntology":null},{"url":"/paper/subequivariant-graph-reinforcement-learning","slug":"subequivariant-graph-reinforcement-learning","title":"Subequivariant Graph Reinforcement Learning in 3D Environments","date":"2023-05-30","arxiv_id":"2305.18951","repositories_listed":1,"syntology":{"n":18,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":13,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/subequivariant-graph-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2305.18951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18951"}},"official":{"repos":["alpc91/sgrl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":13,"ran_from_kinds":["official"]}}},{"url":"/paper/temporally-layered-architecture-for-efficient","slug":"temporally-layered-architecture-for-efficient","title":"Optimizing Attention and Cognitive Control Costs Using Temporally-Layered Architectures","date":"2023-05-30","arxiv_id":"2305.18701","repositories_listed":1,"syntology":null},{"url":"/paper/is-centralized-training-with-decentralized","slug":"is-centralized-training-with-decentralized","title":"Is Centralized Training with Decentralized Execution Framework Centralized Enough for MARL?","date":"2023-05-27","arxiv_id":"2305.17352","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/is-centralized-training-with-decentralized#ran","syntology_url":"https://syntology.ai/paper/2305.17352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17352"}},"official":{"repos":["zyh1999/cadp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/query-policy-misalignment-in-preference-based","slug":"query-policy-misalignment-in-preference-based","title":"Query-Policy Misalignment in Preference-Based Reinforcement Learning","date":"2023-05-27","arxiv_id":"2305.17400","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":1,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/query-policy-misalignment-in-preference-based#ran","syntology_url":"https://syntology.ai/paper/2305.17400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17400"}},"official":{"repos":["huxiao09/qpa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-hierarchical-approach-to-population","slug":"a-hierarchical-approach-to-population","title":"A Hierarchical Approach to Population Training for Human-AI Collaboration","date":"2023-05-26","arxiv_id":"2305.16708","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-pd-control-using-deep-reinforcement","slug":"adaptive-pd-control-using-deep-reinforcement","title":"Adaptive PD Control using Deep Reinforcement Learning for Local-Remote Teleoperation with Stochastic Time Delays","date":"2023-05-26","arxiv_id":"2305.16979","repositories_listed":1,"syntology":null},{"url":"/paper/physical-deep-reinforcement-learning-safety","slug":"physical-deep-reinforcement-learning-safety","title":"Physics-Regulated Deep Reinforcement Learning: Invariant Embeddings","date":"2023-05-26","arxiv_id":"2305.16614","repositories_listed":1,"syntology":null},{"url":"/paper/generating-synergistic-formulaic-alpha","slug":"generating-synergistic-formulaic-alpha","title":"Generating Synergistic Formulaic Alpha Collections via Reinforcement Learning","date":"2023-05-25","arxiv_id":"2306.12964","repositories_listed":1,"syntology":null},{"url":"/paper/learning-safety-constraints-from","slug":"learning-safety-constraints-from","title":"Learning Safety Constraints from Demonstrations with Unknown Rewards","date":"2023-05-25","arxiv_id":"2305.16147","repositories_listed":1,"syntology":null},{"url":"/paper/market-making-with-deep-reinforcement","slug":"market-making-with-deep-reinforcement","title":"Market Making with Deep Reinforcement Learning from Limit Order Books","date":"2023-05-25","arxiv_id":"2305.15821","repositories_listed":1,"syntology":null},{"url":"/paper/proto-iterative-policy-regularized-offline-to","slug":"proto-iterative-policy-regularized-offline-to","title":"PROTO: Iterative Policy Regularized Offline-to-Online Reinforcement Learning","date":"2023-05-25","arxiv_id":"2305.15669","repositories_listed":1,"syntology":null},{"url":"/paper/reward-machine-guided-self-paced","slug":"reward-machine-guided-self-paced","title":"Reward-Machine-Guided, Self-Paced Reinforcement Learning","date":"2023-05-25","arxiv_id":"2305.16505","repositories_listed":1,"syntology":null},{"url":"/paper/improving-language-models-with-advantage","slug":"improving-language-models-with-advantage","title":"Leftover Lunch: Advantage-based Offline Reinforcement Learning for Language Models","date":"2023-05-24","arxiv_id":"2305.14718","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/improving-language-models-with-advantage#ran","syntology_url":"https://syntology.ai/paper/2305.14718","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14718"}},"official":{"repos":["abaheti95/lol-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/2305-14550","slug":"2305-14550","title":"When should we prefer Decision Transformers for Offline Reinforcement Learning?","date":"2023-05-23","arxiv_id":"2305.14550","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/2305-14550#ran","syntology_url":"https://syntology.ai/paper/2305.14550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14550"}},"official":{"repos":["prajjwal1/rl_paradigm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/conditional-mutual-information-for-1","slug":"conditional-mutual-information-for-1","title":"Conditional Mutual Information for Disentangled Representations in Reinforcement Learning","date":"2023-05-23","arxiv_id":"2305.14133","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conditional-mutual-information-for-1#ran","syntology_url":"https://syntology.ai/paper/2305.14133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14133"}},"official":{"repos":["uoe-agents/cmid"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/constrained-reinforcement-learning-for-4","slug":"constrained-reinforcement-learning-for-4","title":"Constrained Reinforcement Learning for Dynamic Material Handling","date":"2023-05-23","arxiv_id":"2305.13824","repositories_listed":1,"syntology":null},{"url":"/paper/guard-a-safe-reinforcement-learning-benchmark","slug":"guard-a-safe-reinforcement-learning-benchmark","title":"GUARD: A Safe Reinforcement Learning Benchmark","date":"2023-05-23","arxiv_id":"2305.13681","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/guard-a-safe-reinforcement-learning-benchmark#ran","syntology_url":"https://syntology.ai/paper/2305.13681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13681"}},"official":{"repos":["intelligent-control-lab/guard"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/xroute-environment-a-novel-reinforcement","slug":"xroute-environment-a-novel-reinforcement","title":"XRoute Environment: A Novel Reinforcement Learning Environment for Routing","date":"2023-05-23","arxiv_id":"2305.13823","repositories_listed":1,"syntology":null},{"url":"/paper/know-your-enemy-investigating-monte-carlo","slug":"know-your-enemy-investigating-monte-carlo","title":"Know your Enemy: Investigating Monte-Carlo Tree Search with Opponent Models in Pommerman","date":"2023-05-22","arxiv_id":"2305.13206","repositories_listed":1,"syntology":null},{"url":"/paper/multi-task-hierarchical-adversarial-inverse","slug":"multi-task-hierarchical-adversarial-inverse","title":"Multi-task Hierarchical Adversarial Inverse Reinforcement Learning","date":"2023-05-22","arxiv_id":"2305.12633","repositories_listed":1,"syntology":null},{"url":"/paper/policy-representation-via-diffusion","slug":"policy-representation-via-diffusion","title":"Policy Representation via Diffusion Probability Model for Reinforcement Learning","date":"2023-05-22","arxiv_id":"2305.13122","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/policy-representation-via-diffusion#ran","syntology_url":"https://syntology.ai/paper/2305.13122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13122"}},"official":{"repos":["bellmantimehut/dipo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/road-planning-for-slums-via-deep","slug":"road-planning-for-slums-via-deep","title":"Road Planning for Slums via Deep Reinforcement Learning","date":"2023-05-22","arxiv_id":"2305.13060","repositories_listed":1,"syntology":null},{"url":"/paper/testing-of-deep-reinforcement-learning-agents","slug":"testing-of-deep-reinforcement-learning-agents","title":"Testing of Deep Reinforcement Learning Agents with Surrogate Models","date":"2023-05-22","arxiv_id":"2305.12751","repositories_listed":1,"syntology":null},{"url":"/paper/bertrlfuzzer-a-bert-and-reinforcement","slug":"bertrlfuzzer-a-bert-and-reinforcement","title":"BertRLFuzzer: A BERT and Reinforcement Learning Based Fuzzer","date":"2023-05-21","arxiv_id":"2305.12534","repositories_listed":1,"syntology":null},{"url":"/paper/learning-diverse-risk-preferences-in","slug":"learning-diverse-risk-preferences-in","title":"Learning Diverse Risk Preferences in Population-based Self-play","date":"2023-05-19","arxiv_id":"2305.11476","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-diverse-risk-preferences-in#ran","syntology_url":"https://syntology.ai/paper/2305.11476","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11476"}},"official":{"repos":["jackory/rpbt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/contrastive-state-augmentations-for","slug":"contrastive-state-augmentations-for","title":"Contrastive State Augmentations for Reinforcement Learning-Based Recommender Systems","date":"2023-05-18","arxiv_id":"2305.11081","repositories_listed":1,"syntology":null},{"url":"/paper/massively-scalable-inverse-reinforcement","slug":"massively-scalable-inverse-reinforcement","title":"Massively Scalable Inverse Reinforcement Learning in Google Maps","date":"2023-05-18","arxiv_id":"2305.11290","repositories_listed":1,"syntology":null}],"record_sha256":"ee3655e34270d38b61dddea4500447e2447c3176ee8996a74bfd81c23e9c17c7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}