{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/24","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":24,"pages_in_order":132,"rows_per_page":100,"rows":[2301,2400],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/23","next":"/task/reinforcement-learning/papers/25","papers":[{"url":"/paper/automating-reinforcement-learning-with","slug":"automating-reinforcement-learning-with","title":"Automating Reinforcement Learning with Example-based Resets","date":"2022-04-05","arxiv_id":"2204.02041","repositories_listed":1,"syntology":null},{"url":"/paper/jump-start-reinforcement-learning","slug":"jump-start-reinforcement-learning","title":"Jump-Start Reinforcement Learning","date":"2022-04-05","arxiv_id":"2204.02372","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-distributed-reinforcement","slug":"multi-agent-distributed-reinforcement","title":"Multi-Agent Distributed Reinforcement Learning for Making Decentralized Offloading Decisions","date":"2022-04-05","arxiv_id":"2204.02267","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-agents-in-colonel","slug":"reinforcement-learning-agents-in-colonel","title":"Reinforcement Learning Agents in Colonel Blotto","date":"2022-04-04","arxiv_id":"2204.02785","repositories_listed":1,"syntology":null},{"url":"/paper/value-gradient-weighted-model-based-1","slug":"value-gradient-weighted-model-based-1","title":"Value Gradient weighted Model-Based Reinforcement Learning","date":"2022-04-04","arxiv_id":"2204.01464","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-guided-by-provable","slug":"reinforcement-learning-guided-by-provable","title":"Reinforcement Learning Guided by Provable Normative Compliance","date":"2022-03-30","arxiv_id":"2203.16275","repositories_listed":1,"syntology":null},{"url":"/paper/text-driven-video-acceleration-a-weakly","slug":"text-driven-video-acceleration-a-weakly","title":"Text-Driven Video Acceleration: A Weakly-Supervised Reinforcement Learning Method","date":"2022-03-29","arxiv_id":"2203.15778","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-risk-tendency-nano-drone-navigation","slug":"adaptive-risk-tendency-nano-drone-navigation","title":"Adaptive Risk-Tendency: Nano Drone Navigation in Cluttered Environments with Distributional Reinforcement Learning","date":"2022-03-28","arxiv_id":"2203.14749","repositories_listed":1,"syntology":null},{"url":"/paper/an-optical-controlling-environment-and","slug":"an-optical-controlling-environment-and","title":"An Optical Control Environment for Benchmarking Reinforcement Learning Algorithms","date":"2022-03-23","arxiv_id":"2203.12114","repositories_listed":1,"syntology":null},{"url":"/paper/asynchronous-reinforcement-learning-for-real","slug":"asynchronous-reinforcement-learning-for-real","title":"Asynchronous Reinforcement Learning for Real-Time Control of Physical Robots","date":"2022-03-23","arxiv_id":"2203.12759","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/asynchronous-reinforcement-learning-for-real#ran","syntology_url":"https://syntology.ai/paper/2203.12759","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.12759"}},"official":{"repos":["yufengyuan/ur5_async_rl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/possibility-before-utility-learning-and-using-1","slug":"possibility-before-utility-learning-and-using-1","title":"Possibility Before Utility: Learning And Using Hierarchical Affordances","date":"2022-03-23","arxiv_id":"2203.12686","repositories_listed":1,"syntology":null},{"url":"/paper/long-short-term-memory-for-spatial-encoding","slug":"long-short-term-memory-for-spatial-encoding","title":"Long Short-Term Memory for Spatial Encoding in Multi-Agent Path Planning","date":"2022-03-21","arxiv_id":"2203.10823","repositories_listed":1,"syntology":null},{"url":"/paper/reccover-detecting-causal-confusion-for","slug":"reccover-detecting-causal-confusion-for","title":"ReCCoVER: Detecting Causal Confusion for Explainable Reinforcement Learning","date":"2022-03-21","arxiv_id":"2203.11211","repositories_listed":1,"syntology":null},{"url":"/paper/microracer-a-didactic-environment-for-deep","slug":"microracer-a-didactic-environment-for-deep","title":"MicroRacer: a didactic environment for Deep Reinforcement Learning","date":"2022-03-20","arxiv_id":"2203.10494","repositories_listed":1,"syntology":null},{"url":"/paper/perceiving-the-world-question-guided","slug":"perceiving-the-world-question-guided","title":"Perceiving the World: Question-guided Reinforcement Learning for Text-based Games","date":"2022-03-20","arxiv_id":"2204.09597","repositories_listed":1,"syntology":null},{"url":"/paper/quantum-multi-agent-reinforcement-learning","slug":"quantum-multi-agent-reinforcement-learning","title":"Quantum Multi-Agent Reinforcement Learning via Variational Quantum Circuit Design","date":"2022-03-20","arxiv_id":"2203.10443","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-automatic","slug":"reinforcement-learning-for-automatic","title":"Reinforcement learning for automatic quadrilateral mesh generation: a soft actor-critic approach","date":"2022-03-19","arxiv_id":"2203.11203","repositories_listed":1,"syntology":null},{"url":"/paper/teachable-reinforcement-learning-via-advice-1","slug":"teachable-reinforcement-learning-via-advice-1","title":"Teachable Reinforcement Learning via Advice Distillation","date":"2022-03-19","arxiv_id":"2203.11197","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/teachable-reinforcement-learning-via-advice-1#ran","syntology_url":"https://syntology.ai/paper/2203.11197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.11197"}},"official":{"repos":["rll-research/teachable"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/semi-markov-offline-reinforcement-learning","slug":"semi-markov-offline-reinforcement-learning","title":"Semi-Markov Offline Reinforcement Learning for Healthcare","date":"2022-03-17","arxiv_id":"2203.09365","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/semi-markov-offline-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2203.09365","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.09365"}},"official":{"repos":["mary-wu/smdp"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/coach-assisted-multi-agent-reinforcement","slug":"coach-assisted-multi-agent-reinforcement","title":"Coach-assisted Multi-Agent Reinforcement Learning Framework for Unexpected Crashed Agents","date":"2022-03-16","arxiv_id":"2203.08454","repositories_listed":1,"syntology":null},{"url":"/paper/latent-variable-advantage-weighted-policy","slug":"latent-variable-advantage-weighted-policy","title":"Latent-Variable Advantage-Weighted Policy Optimization for Offline RL","date":"2022-03-16","arxiv_id":"2203.08949","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/latent-variable-advantage-weighted-policy#ran","syntology_url":"https://syntology.ai/paper/2203.08949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08949"}},"official":null}},{"url":"/paper/zipfian-environments-for-reinforcement","slug":"zipfian-environments-for-reinforcement","title":"Zipfian environments for Reinforcement Learning","date":"2022-03-15","arxiv_id":"2203.08222","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/zipfian-environments-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2203.08222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08222"}},"official":{"repos":["deepmind/zipfian_environments"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/l2explorer-a-lifelong-reinforcement-learning","slug":"l2explorer-a-lifelong-reinforcement-learning","title":"L2Explorer: A Lifelong Reinforcement Learning Assessment Environment","date":"2022-03-14","arxiv_id":"2203.07454","repositories_listed":1,"syntology":null},{"url":"/paper/orchestrated-value-mapping-for-reinforcement-1","slug":"orchestrated-value-mapping-for-reinforcement-1","title":"Orchestrated Value Mapping for Reinforcement Learning","date":"2022-03-14","arxiv_id":"2203.07171","repositories_listed":1,"syntology":null},{"url":"/paper/curriculum-based-reinforcement-learning-for","slug":"curriculum-based-reinforcement-learning-for","title":"Curriculum-based Reinforcement Learning for Distribution System Critical Load Restoration","date":"2022-03-08","arxiv_id":"2203.04166","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-entity-1","slug":"deep-reinforcement-learning-for-entity-1","title":"Deep Reinforcement Learning for Entity Alignment","date":"2022-03-07","arxiv_id":"2203.03315","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-for-entity-1#ran","syntology_url":"https://syntology.ai/paper/2203.03315","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.03315"}},"official":{"repos":["guolingbing/rlea"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/influencing-long-term-behavior-in-multiagent","slug":"influencing-long-term-behavior-in-multiagent","title":"Influencing Long-Term Behavior in Multiagent Reinforcement Learning","date":"2022-03-07","arxiv_id":"2203.03535","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":3,"n_ran_checked":4,"n_instrument":6,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"10 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/influencing-long-term-behavior-in-multiagent#ran","syntology_url":"https://syntology.ai/paper/2203.03535","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.03535"}},"official":{"repos":["dkkim93/further"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/on-credit-assignment-in-hierarchical","slug":"on-credit-assignment-in-hierarchical","title":"On Credit Assignment in Hierarchical Reinforcement Learning","date":"2022-03-07","arxiv_id":"2203.03292","repositories_listed":1,"syntology":null},{"url":"/paper/on-practical-reinforcement-learning-provable","slug":"on-practical-reinforcement-learning-provable","title":"On Practical Reinforcement Learning: Provable Robustness, Scalability, and Statistical Efficiency","date":"2022-03-03","arxiv_id":"2203.01758","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-in-possibly","slug":"reinforcement-learning-in-possibly","title":"Testing Stationarity and Change Point Detection in Reinforcement Learning","date":"2022-03-03","arxiv_id":"2203.01707","repositories_listed":1,"syntology":null},{"url":"/paper/andes-gym-a-versatile-environment-for-deep","slug":"andes-gym-a-versatile-environment-for-deep","title":"Andes_gym: A Versatile Environment for Deep Reinforcement Learning in Power Systems","date":"2022-03-02","arxiv_id":"2203.01292","repositories_listed":1,"syntology":null},{"url":"/paper/ai-planning-annotation-for-sample-efficient","slug":"ai-planning-annotation-for-sample-efficient","title":"Hierarchical Reinforcement Learning with AI Planning Models","date":"2022-03-01","arxiv_id":"2203.00669","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-generalization-of-representations-in","slug":"on-the-generalization-of-representations-in","title":"On the Generalization of Representations in Reinforcement Learning","date":"2022-03-01","arxiv_id":"2203.00543","repositories_listed":1,"syntology":null},{"url":"/paper/avalanche-rl-a-continual-reinforcement","slug":"avalanche-rl-a-continual-reinforcement","title":"Avalanche RL: a Continual Reinforcement Learning Library","date":"2022-02-28","arxiv_id":"2202.13657","repositories_listed":1,"syntology":null},{"url":"/paper/rl-pgo-reinforcement-learning-based-planar","slug":"rl-pgo-reinforcement-learning-based-planar","title":"RL-PGO: Reinforcement Learning-based Planar Pose-Graph Optimization","date":"2022-02-26","arxiv_id":"2202.13221","repositories_listed":1,"syntology":null},{"url":"/paper/building-a-3-player-mahjong-ai-using-deep","slug":"building-a-3-player-mahjong-ai-using-deep","title":"Building a 3-Player Mahjong AI using Deep Reinforcement Learning","date":"2022-02-25","arxiv_id":"2202.12847","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-with-sticky-mittens-reinforcement","slug":"exploring-with-sticky-mittens-reinforcement","title":"Exploring with Sticky Mittens: Reinforcement Learning with Expert Interventions via Option Templates","date":"2022-02-25","arxiv_id":"2202.12967","repositories_listed":1,"syntology":null},{"url":"/paper/pessimistic-bootstrapping-for-uncertainty-1","slug":"pessimistic-bootstrapping-for-uncertainty-1","title":"Pessimistic Bootstrapping for Uncertainty-Driven Offline Reinforcement Learning","date":"2022-02-23","arxiv_id":"2202.11566","repositories_listed":1,"syntology":null},{"url":"/paper/a-comparative-study-of-deep-reinforcement","slug":"a-comparative-study-of-deep-reinforcement","title":"A Comparative Study of Deep Reinforcement Learning-based Transferable Energy Management Strategies for Hybrid Electric Vehicles","date":"2022-02-22","arxiv_id":"2202.11514","repositories_listed":1,"syntology":null},{"url":"/paper/shaping-advice-in-deep-reinforcement-learning","slug":"shaping-advice-in-deep-reinforcement-learning","title":"Shaping Advice in Deep Reinforcement Learning","date":"2022-02-19","arxiv_id":"2202.09489","repositories_listed":1,"syntology":null},{"url":"/paper/darl1n-distributed-multi-agent-reinforcement","slug":"darl1n-distributed-multi-agent-reinforcement","title":"Distributed Multi-Agent Reinforcement Learning with One-hop Neighbors and Compute Straggler Mitigation","date":"2022-02-18","arxiv_id":"2202.09019","repositories_listed":1,"syntology":null},{"url":"/paper/vrl3-a-data-driven-framework-for-visual-deep","slug":"vrl3-a-data-driven-framework-for-visual-deep","title":"VRL3: A Data-Driven Framework for Visual Deep Reinforcement Learning","date":"2022-02-17","arxiv_id":"2202.10324","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":4,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/vrl3-a-data-driven-framework-for-visual-deep#ran","syntology_url":"https://syntology.ai/paper/2202.10324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.10324"}},"official":{"repos":["facebookresearch/drqv2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/open-ended-reinforcement-learning-with-neural","slug":"open-ended-reinforcement-learning-with-neural","title":"Open-Ended Reinforcement Learning with Neural Reward Functions","date":"2022-02-16","arxiv_id":"2202.08266","repositories_listed":1,"syntology":null},{"url":"/paper/cup-a-conservative-update-policy-algorithm-1","slug":"cup-a-conservative-update-policy-algorithm-1","title":"CUP: A Conservative Update Policy Algorithm for Safe Reinforcement Learning","date":"2022-02-15","arxiv_id":"2202.07565","repositories_listed":1,"syntology":null},{"url":"/paper/energy-efficient-parking-analytics-system","slug":"energy-efficient-parking-analytics-system","title":"Energy-Efficient Parking Analytics System using Deep Reinforcement Learning","date":"2022-02-15","arxiv_id":"2202.08973","repositories_listed":1,"syntology":null},{"url":"/paper/graph-meta-reinforcement-learning-for","slug":"graph-meta-reinforcement-learning-for","title":"Graph Meta-Reinforcement Learning for Transferable Autonomous Mobility-on-Demand","date":"2022-02-15","arxiv_id":"2202.07147","repositories_listed":1,"syntology":null},{"url":"/paper/safe-reinforcement-learning-by-imagining-the-1","slug":"safe-reinforcement-learning-by-imagining-the-1","title":"Safe Reinforcement Learning by Imagining the Near Future","date":"2022-02-15","arxiv_id":"2202.07789","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safe-reinforcement-learning-by-imagining-the-1#ran","syntology_url":"https://syntology.ai/paper/2202.07789","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.07789"}},"official":{"repos":["gwthomas/safe-mbpo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-reward-models-for-cooperative","slug":"learning-reward-models-for-cooperative","title":"Learning Reward Models for Cooperative Trajectory Planning with Inverse Reinforcement Learning and Monte Carlo Tree Search","date":"2022-02-14","arxiv_id":"2202.06443","repositories_listed":1,"syntology":null},{"url":"/paper/saute-rl-almost-surely-safe-reinforcement","slug":"saute-rl-almost-surely-safe-reinforcement","title":"Saute RL: Almost Surely Safe Reinforcement Learning Using State Augmentation","date":"2022-02-14","arxiv_id":"2202.06558","repositories_listed":1,"syntology":null},{"url":"/paper/goal-recognition-as-reinforcement-learning","slug":"goal-recognition-as-reinforcement-learning","title":"Goal Recognition as Reinforcement Learning","date":"2022-02-13","arxiv_id":"2202.06356","repositories_listed":1,"syntology":null},{"url":"/paper/uncovering-instabilities-in-variational","slug":"uncovering-instabilities-in-variational","title":"Uncovering Instabilities in Variational-Quantum Deep Q-Networks","date":"2022-02-10","arxiv_id":"2202.05195","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-nonparametrics-for-offline-skill","slug":"bayesian-nonparametrics-for-offline-skill","title":"Bayesian Nonparametrics for Offline Skill Discovery","date":"2022-02-09","arxiv_id":"2202.04675","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bayesian-nonparametrics-for-offline-skill#ran","syntology_url":"https://syntology.ai/paper/2202.04675","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.04675"}},"official":{"repos":["layer6ai-labs/bnpo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/contextualize-me-the-case-for-context-in","slug":"contextualize-me-the-case-for-context-in","title":"Contextualize Me -- The Case for Context in Reinforcement Learning","date":"2022-02-09","arxiv_id":"2202.04500","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/contextualize-me-the-case-for-context-in#ran","syntology_url":"https://syntology.ai/paper/2202.04500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.04500"}},"official":{"repos":["automl/CARL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-sparse-rewards-1","slug":"reinforcement-learning-with-sparse-rewards-1","title":"Reinforcement Learning with Sparse Rewards using Guidance from Offline Demonstration","date":"2022-02-09","arxiv_id":"2202.04628","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-sparse-rewards-1#ran","syntology_url":"https://syntology.ai/paper/2202.04628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.04628"}},"official":{"repos":["desikrengarajan/logo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/approximating-gradients-for-differentiable","slug":"approximating-gradients-for-differentiable","title":"Approximating Gradients for Differentiable Quality Diversity in Reinforcement Learning","date":"2022-02-08","arxiv_id":"2202.03666","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/approximating-gradients-for-differentiable#ran","syntology_url":"https://syntology.ai/paper/2202.03666","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.03666"}},"official":{"repos":["icaros-usc/dqd-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bingham-policy-parameterization-for-3d","slug":"bingham-policy-parameterization-for-3d","title":"Bingham Policy Parameterization for 3D Rotations in Reinforcement Learning","date":"2022-02-08","arxiv_id":"2202.03957","repositories_listed":1,"syntology":null},{"url":"/paper/skrl-modular-and-flexible-library-for","slug":"skrl-modular-and-flexible-library-for","title":"skrl: Modular and Flexible Library for Reinforcement Learning","date":"2022-02-08","arxiv_id":"2202.03825","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-warfarin-dosing-using-deep","slug":"optimizing-warfarin-dosing-using-deep","title":"Optimizing Warfarin Dosing using Deep Reinforcement Learning","date":"2022-02-07","arxiv_id":"2202.03486","repositories_listed":1,"syntology":null},{"url":"/paper/learning-synthetic-environments-and-reward-1","slug":"learning-synthetic-environments-and-reward-1","title":"Learning Synthetic Environments and Reward Networks for Reinforcement Learning","date":"2022-02-06","arxiv_id":"2202.02790","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-synthetic-environments-and-reward-1#ran","syntology_url":"https://syntology.ai/paper/2202.02790","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.02790"}},"official":{"repos":["automl/learning_environments"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-approximate-symbolic-models-for","slug":"leveraging-approximate-symbolic-models-for","title":"Leveraging Approximate Symbolic Models for Reinforcement Learning via Skill Diversity","date":"2022-02-06","arxiv_id":"2202.02886","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-multi-item","slug":"reinforcement-learning-for-multi-item","title":"Reinforcement learning for multi-item retrieval in the puzzle-based storage system","date":"2022-02-05","arxiv_id":"2202.03424","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-sequential-experimental-design","slug":"optimizing-sequential-experimental-design","title":"Optimizing Sequential Experimental Design with Deep Reinforcement Learning","date":"2022-02-02","arxiv_id":"2202.00821","repositories_listed":1,"syntology":null},{"url":"/paper/distributional-reinforcement-learning-via","slug":"distributional-reinforcement-learning-via","title":"Distributional Reinforcement Learning with Regularized Wasserstein Loss","date":"2022-02-01","arxiv_id":"2202.00769","repositories_listed":1,"syntology":null},{"url":"/paper/tutorial-on-amortized-optimization-for","slug":"tutorial-on-amortized-optimization-for","title":"Tutorial on amortized optimization","date":"2022-02-01","arxiv_id":"2202.00665","repositories_listed":1,"syntology":null},{"url":"/paper/bellman-meets-hawkes-model-based","slug":"bellman-meets-hawkes-model-based","title":"Bellman Meets Hawkes: Model-Based Reinforcement Learning via Temporal Point Processes","date":"2022-01-29","arxiv_id":"2201.12569","repositories_listed":1,"syntology":null},{"url":"/paper/explaining-reinforcement-learning-policies","slug":"explaining-reinforcement-learning-policies","title":"Explaining Reinforcement Learning Policies through Counterfactual Trajectories","date":"2022-01-29","arxiv_id":"2201.12462","repositories_listed":1,"syntology":{"n":11,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":11,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/explaining-reinforcement-learning-policies#ran","syntology_url":"https://syntology.ai/paper/2201.12462","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12462"}},"official":{"repos":["juliusfrost/cfrl-rllib"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/can-wikipedia-help-offline-reinforcement","slug":"can-wikipedia-help-offline-reinforcement","title":"Can Wikipedia Help Offline Reinforcement Learning?","date":"2022-01-28","arxiv_id":"2201.12122","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-wikipedia-help-offline-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2201.12122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12122"}},"official":{"repos":["machelreid/can-wikipedia-help-offline-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mask-based-latent-reconstruction-for","slug":"mask-based-latent-reconstruction-for","title":"Mask-based Latent Reconstruction for Reinforcement Learning","date":"2022-01-28","arxiv_id":"2201.12096","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":4,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 4 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mask-based-latent-reconstruction-for#ran","syntology_url":"https://syntology.ai/paper/2201.12096","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12096"}},"official":{"repos":["microsoft/Mask-based-Latent-Reconstruction"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/towards-safe-reinforcement-learning-with-a","slug":"towards-safe-reinforcement-learning-with-a","title":"Towards Safe Reinforcement Learning with a Safety Editor Policy","date":"2022-01-28","arxiv_id":"2201.12427","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-safe-reinforcement-learning-with-a#ran","syntology_url":"https://syntology.ai/paper/2201.12427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12427"}},"official":{"repos":["hnyu/seditor"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/exploration-with-a-finite-brain","slug":"exploration-with-a-finite-brain","title":"Modeling Human Exploration Through Resource-Rational Reinforcement Learning","date":"2022-01-27","arxiv_id":"2201.11817","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":3,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exploration-with-a-finite-brain#ran","syntology_url":"https://syntology.ai/paper/2201.11817","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.11817"}},"official":{"repos":["marcelbinz/resource-rational-reinforcement-learning"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-learning-dynamics-in-rl-using","slug":"rethinking-learning-dynamics-in-rl-using","title":"Boosting Exploration in Multi-Task Reinforcement Learning using Adversarial Networks","date":"2022-01-27","arxiv_id":"2201.11783","repositories_listed":1,"syntology":null},{"url":"/paper/moolib-a-platform-for-distributed-rl","slug":"moolib-a-platform-for-distributed-rl","title":"moolib: A Platform for Distributed RL","date":"2022-01-26","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/generative-planning-for-temporally-1","slug":"generative-planning-for-temporally-1","title":"Generative Planning for Temporally Coordinated Exploration in Reinforcement Learning","date":"2022-01-24","arxiv_id":"2201.09765","repositories_listed":1,"syntology":null},{"url":"/paper/pearl-parallel-evolutionary-and-reinforcement","slug":"pearl-parallel-evolutionary-and-reinforcement","title":"Pearl: Parallel Evolutionary and Reinforcement Learning Library","date":"2022-01-24","arxiv_id":"2201.09568","repositories_listed":1,"syntology":null},{"url":"/paper/the-paradox-of-choice-using-attention-in","slug":"the-paradox-of-choice-using-attention-in","title":"The Paradox of Choice: Using Attention in Hierarchical Reinforcement Learning","date":"2022-01-24","arxiv_id":"2201.09653","repositories_listed":1,"syntology":null},{"url":"/paper/bag-of-tricks-for-natural-policy-gradient","slug":"bag-of-tricks-for-natural-policy-gradient","title":"Understanding the Effects of Second-Order Approximations in Natural Policy Gradient Reinforcement Learning","date":"2022-01-22","arxiv_id":"2201.09104","repositories_listed":1,"syntology":null},{"url":"/paper/environment-generation-for-zero-shot-1","slug":"environment-generation-for-zero-shot-1","title":"Environment Generation for Zero-Shot Compositional Reinforcement Learning","date":"2022-01-21","arxiv_id":"2201.08896","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/environment-generation-for-zero-shot-1#ran","syntology_url":"https://syntology.ai/paper/2201.08896","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.08896"}},"official":{"repos":["google-research/google-research"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/goal-conditioned-reinforcement-learning","slug":"goal-conditioned-reinforcement-learning","title":"Goal-Conditioned Reinforcement Learning: Problems and Solutions","date":"2022-01-20","arxiv_id":"2201.08299","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-textbook","slug":"reinforcement-learning-textbook","title":"Reinforcement Learning Textbook","date":"2022-01-19","arxiv_id":"2201.09746","repositories_listed":1,"syntology":null},{"url":"/paper/smart-magnetic-microrobots-learn-to-swim-with","slug":"smart-magnetic-microrobots-learn-to-swim-with","title":"Smart Magnetic Microrobots Learn to Swim with Deep Reinforcement Learning","date":"2022-01-14","arxiv_id":"2201.05599","repositories_listed":1,"syntology":null},{"url":"/paper/weakly-supervised-scene-text-detection-using","slug":"weakly-supervised-scene-text-detection-using","title":"Weakly Supervised Scene Text Detection using Deep Reinforcement Learning","date":"2022-01-13","arxiv_id":"2201.04866","repositories_listed":1,"syntology":null},{"url":"/paper/verified-probabilistic-policies-for-deep","slug":"verified-probabilistic-policies-for-deep","title":"Verified Probabilistic Policies for Deep Reinforcement Learning","date":"2022-01-10","arxiv_id":"2201.03698","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-deep-reinforcement-learning-5","slug":"sample-efficient-deep-reinforcement-learning-5","title":"Sample Efficient Deep Reinforcement Learning via Uncertainty Estimation","date":"2022-01-05","arxiv_id":"2201.01666","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sample-efficient-deep-reinforcement-learning-5#ran","syntology_url":"https://syntology.ai/paper/2201.01666","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.01666"}},"official":{"repos":["montrealrobotics/iv_rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/constraint-sampling-reinforcement-learning","slug":"constraint-sampling-reinforcement-learning","title":"Constraint Sampling Reinforcement Learning: Incorporating Expertise For Faster Learning","date":"2021-12-30","arxiv_id":"2112.15221","repositories_listed":1,"syntology":null},{"url":"/paper/sequential-episodic-control","slug":"sequential-episodic-control","title":"Sequential memory improves sample and memory efficiency in Episodic Control","date":"2021-12-29","arxiv_id":"2112.14734","repositories_listed":1,"syntology":null},{"url":"/paper/exponential-family-model-based-reinforcement","slug":"exponential-family-model-based-reinforcement","title":"Exponential Family Model-Based Reinforcement Learning via Score Matching","date":"2021-12-28","arxiv_id":"2112.14195","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/exponential-family-model-based-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2112.14195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.14195"}},"official":{"repos":["anmolkabra/score-matching-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/intelligent-traffic-light-via-policy-based","slug":"intelligent-traffic-light-via-policy-based","title":"Intelligent Traffic Light via Policy-based Deep Reinforcement Learning","date":"2021-12-27","arxiv_id":"2112.13817","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-dynamic-convex","slug":"reinforcement-learning-with-dynamic-convex","title":"Reinforcement Learning with Dynamic Convex Risk Measures","date":"2021-12-26","arxiv_id":"2112.13414","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-dynamic-convex#ran","syntology_url":"https://syntology.ai/paper/2112.13414","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.13414"}},"official":{"repos":["acoache/rl-dynamicconvexrisk"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-walk-with-dual-agents-for","slug":"learning-to-walk-with-dual-agents-for","title":"Learning to Walk with Dual Agents for Knowledge Graph Reasoning","date":"2021-12-23","arxiv_id":"2112.12876","repositories_listed":1,"syntology":null},{"url":"/paper/safety-and-liveness-guarantees-through-reach","slug":"safety-and-liveness-guarantees-through-reach","title":"Safety and Liveness Guarantees through Reach-Avoid Reinforcement Learning","date":"2021-12-23","arxiv_id":"2112.12288","repositories_listed":1,"syntology":null},{"url":"/paper/alpha-mini-minichess-agent-with-deep","slug":"alpha-mini-minichess-agent-with-deep","title":"Alpha-Mini: Minichess Agent with Deep Reinforcement Learning","date":"2021-12-22","arxiv_id":"2112.13666","repositories_listed":1,"syntology":null},{"url":"/paper/direct-behavior-specification-via-constrained","slug":"direct-behavior-specification-via-constrained","title":"Direct Behavior Specification via Constrained Reinforcement Learning","date":"2021-12-22","arxiv_id":"2112.12228","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-the-robustness-of-deep","slug":"evaluating-the-robustness-of-deep","title":"Evaluating the Robustness of Deep Reinforcement Learning for Autonomous Policies in a Multi-agent Urban Driving Environment","date":"2021-12-22","arxiv_id":"2112.11947","repositories_listed":1,"syntology":null},{"url":"/paper/newsvendor-model-with-deep-reinforcement","slug":"newsvendor-model-with-deep-reinforcement","title":"Newsvendor Model with Deep Reinforcement Learning","date":"2021-12-22","arxiv_id":"2112.12544","repositories_listed":1,"syntology":null},{"url":"/paper/variational-quantum-soft-actor-critic","slug":"variational-quantum-soft-actor-critic","title":"Variational Quantum Soft Actor-Critic","date":"2021-12-20","arxiv_id":"2112.11921","repositories_listed":1,"syntology":null},{"url":"/paper/space-non-cooperative-object-active-tracking","slug":"space-non-cooperative-object-active-tracking","title":"Space Non-cooperative Object Active Tracking with Deep Reinforcement Learning","date":"2021-12-18","arxiv_id":"2112.09854","repositories_listed":1,"syntology":null},{"url":"/paper/inherently-explainable-reinforcement-learning","slug":"inherently-explainable-reinforcement-learning","title":"Inherently Explainable Reinforcement Learning in Natural Language","date":"2021-12-16","arxiv_id":"2112.08907","repositories_listed":1,"syntology":null},{"url":"/paper/feature-attending-recurrent-modules-for","slug":"feature-attending-recurrent-modules-for","title":"Feature-Attending Recurrent Modules for Generalization in Reinforcement Learning","date":"2021-12-15","arxiv_id":"2112.08369","repositories_listed":1,"syntology":null},{"url":"/paper/conjugated-discrete-distributions-for","slug":"conjugated-discrete-distributions-for","title":"Conjugated Discrete Distributions for Distributional Reinforcement Learning","date":"2021-12-14","arxiv_id":"2112.07424","repositories_listed":1,"syntology":null},{"url":"/paper/conservative-and-adaptive-penalty-for-model","slug":"conservative-and-adaptive-penalty-for-model","title":"Conservative and Adaptive Penalty for Model-Based Safe Reinforcement Learning","date":"2021-12-14","arxiv_id":"2112.07701","repositories_listed":1,"syntology":null}],"record_sha256":"acba78270a24251ce328131cb5a1be58e9eb884cb6ad0c751425d76b89bce219","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}