{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/22","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":22,"pages_in_order":135,"rows_per_page":100,"rows":[2101,2200],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/21","next":"/task/reinforcement-learning-2/papers/23","papers":[{"url":"/paper/language-conditioned-reinforcement-learning","slug":"language-conditioned-reinforcement-learning","title":"Language-Conditioned Reinforcement Learning to Solve Misunderstandings with Action Corrections","date":"2022-11-18","arxiv_id":"2211.10168","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-conditioned-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2211.10168","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.10168"}},"official":{"repos":["frankroeder/lanro-gym"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/provable-defense-against-backdoor-policies-in","slug":"provable-defense-against-backdoor-policies-in","title":"Provable Defense against Backdoor Policies in Reinforcement Learning","date":"2022-11-18","arxiv_id":"2211.10530","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/provable-defense-against-backdoor-policies-in#ran","syntology_url":"https://syntology.ai/paper/2211.10530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.10530"}},"official":{"repos":["skbharti/provable-defense-in-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/agent-state-construction-with-auxiliary","slug":"agent-state-construction-with-auxiliary","title":"Agent-State Construction with Auxiliary Inputs","date":"2022-11-15","arxiv_id":"2211.07805","repositories_listed":1,"syntology":null},{"url":"/paper/explainable-action-advising-for-multi-agent","slug":"explainable-action-advising-for-multi-agent","title":"Explainable Action Advising for Multi-Agent Reinforcement Learning","date":"2022-11-15","arxiv_id":"2211.07882","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchically-structured-task-agnostic","slug":"hierarchically-structured-task-agnostic","title":"Hierarchically Structured Task-Agnostic Continual Learning","date":"2022-11-14","arxiv_id":"2211.07725","repositories_listed":1,"syntology":null},{"url":"/paper/interactively-learning-to-summarise-timelines","slug":"interactively-learning-to-summarise-timelines","title":"Towards Abstractive Timeline Summarisation using Preference-based Reinforcement Learning","date":"2022-11-14","arxiv_id":"2211.07596","repositories_listed":1,"syntology":null},{"url":"/paper/towards-data-driven-offline-simulations-for","slug":"towards-data-driven-offline-simulations-for","title":"Towards Data-Driven Offline Simulations for Online Reinforcement Learning","date":"2022-11-14","arxiv_id":"2211.07614","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-data-driven-offline-simulations-for#ran","syntology_url":"https://syntology.ai/paper/2211.07614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.07614"}},"official":{"repos":["microsoft/rl-offline-simulation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-heterogeneous-agent-cooperation-via","slug":"learning-heterogeneous-agent-cooperation-via","title":"Learning Heterogeneous Agent Cooperation via Multiagent League Training","date":"2022-11-13","arxiv_id":"2211.11616","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-explainable-reinforcement","slug":"a-survey-on-explainable-reinforcement","title":"A Survey on Explainable Reinforcement Learning: Concepts, Algorithms, Challenges","date":"2022-11-12","arxiv_id":"2211.06665","repositories_listed":1,"syntology":null},{"url":"/paper/online-anomalous-subtrajectory-detection-on","slug":"online-anomalous-subtrajectory-detection-on","title":"Online Anomalous Subtrajectory Detection on Road Networks with Deep Reinforcement Learning","date":"2022-11-12","arxiv_id":"2211.08415","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-deep-reinforcement-learning-with-1","slug":"efficient-deep-reinforcement-learning-with-1","title":"Efficient Deep Reinforcement Learning with Predictive Processing Proximal Policy Optimization","date":"2022-11-11","arxiv_id":"2211.06236","repositories_listed":1,"syntology":null},{"url":"/paper/global-and-local-analysis-of-interestingness","slug":"global-and-local-analysis-of-interestingness","title":"Global and Local Analysis of Interestingness for Competency-Aware Deep Reinforcement Learning","date":"2022-11-11","arxiv_id":"2211.06376","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-in-an-adaptable-chess","slug":"reinforcement-learning-in-an-adaptable-chess","title":"Reinforcement Learning in an Adaptable Chess Environment for Detecting Human-understandable Concepts","date":"2022-11-10","arxiv_id":"2211.05500","repositories_listed":1,"syntology":null},{"url":"/paper/deep-w-networks-solving-multi-objective","slug":"deep-w-networks-solving-multi-objective","title":"Deep W-Networks: Solving Multi-Objective Optimisation Problems With Deep Reinforcement Learning","date":"2022-11-09","arxiv_id":"2211.04813","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-sequentiality-in-reinforcement","slug":"leveraging-sequentiality-in-reinforcement","title":"Leveraging Sequentiality in Reinforcement Learning from a Single Demonstration","date":"2022-11-09","arxiv_id":"2211.04786","repositories_listed":1,"syntology":null},{"url":"/paper/doubly-inhomogeneous-reinforcement-learning","slug":"doubly-inhomogeneous-reinforcement-learning","title":"Doubly Inhomogeneous Reinforcement Learning","date":"2022-11-08","arxiv_id":"2211.03983","repositories_listed":1,"syntology":null},{"url":"/paper/curriculum-based-asymmetric-multi-task","slug":"curriculum-based-asymmetric-multi-task","title":"Curriculum-based Asymmetric Multi-task Reinforcement Learning","date":"2022-11-07","arxiv_id":"2211.03352","repositories_listed":1,"syntology":null},{"url":"/paper/design-process-is-a-reinforcement-learning","slug":"design-process-is-a-reinforcement-learning","title":"Design Process is a Reinforcement Learning Problem","date":"2022-11-06","arxiv_id":"2211.03136","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-quality-diversity-algorithms-on","slug":"benchmarking-quality-diversity-algorithms-on","title":"Benchmarking Quality-Diversity Algorithms on Neuroevolution for Reinforcement Learning","date":"2022-11-04","arxiv_id":"2211.02193","repositories_listed":1,"syntology":null},{"url":"/paper/diversity-based-deep-reinforcement-learning","slug":"diversity-based-deep-reinforcement-learning","title":"Diversity-based Deep Reinforcement Learning Towards Multidimensional Difficulty for Fighting Game AI","date":"2022-11-04","arxiv_id":"2211.02759","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diversity-based-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2211.02759","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.02759"}},"official":{"repos":["emily-halina/brisket"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-benefits-of-model-based-generalization-in","slug":"the-benefits-of-model-based-generalization-in","title":"The Benefits of Model-Based Generalization in Reinforcement Learning","date":"2022-11-04","arxiv_id":"2211.02222","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-benefits-of-model-based-generalization-in#ran","syntology_url":"https://syntology.ai/paper/2211.02222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.02222"}},"official":{"repos":["kenjyoung/model_generalization_code_supplement"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-safety-in-model-based-reinforcement","slug":"learning-safety-in-model-based-reinforcement","title":"Learning safety in model-based Reinforcement Learning using MPC and Gaussian Processes","date":"2022-11-03","arxiv_id":"2211.01860","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-fully-observable-policies-for","slug":"leveraging-fully-observable-policies-for","title":"Leveraging Fully Observable Policies for Learning under Partial Observability","date":"2022-11-03","arxiv_id":"2211.01991","repositories_listed":1,"syntology":null},{"url":"/paper/synthesis-of-separation-processes-with","slug":"synthesis-of-separation-processes-with","title":"Synthesis of separation processes with reinforcement learning","date":"2022-11-03","arxiv_id":"2211.04327","repositories_listed":1,"syntology":null},{"url":"/paper/behavior-prior-representation-learning-for","slug":"behavior-prior-representation-learning-for","title":"Behavior Prior Representation learning for Offline Reinforcement Learning","date":"2022-11-02","arxiv_id":"2211.00863","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/behavior-prior-representation-learning-for#ran","syntology_url":"https://syntology.ai/paper/2211.00863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.00863"}},"official":{"repos":["bit1029public/offline_bpr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-reinforcement-learning-for-13","slug":"multi-agent-reinforcement-learning-for-13","title":"Multi-Agent Reinforcement Learning for Adaptive Mesh Refinement","date":"2022-11-02","arxiv_id":"2211.00801","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/multi-agent-reinforcement-learning-for-13#ran","syntology_url":"https://syntology.ai/paper/2211.00801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.00801"}},"official":{"repos":["011235813/marl-amr"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/spatial-temporal-recurrent-reinforcement","slug":"spatial-temporal-recurrent-reinforcement","title":"Spatial-temporal recurrent reinforcement learning for autonomous ships","date":"2022-11-02","arxiv_id":"2211.01004","repositories_listed":1,"syntology":null},{"url":"/paper/can-maker-taker-fees-prevent-algorithmic","slug":"can-maker-taker-fees-prevent-algorithmic","title":"Can maker-taker fees prevent algorithmic cooperation in market making?","date":"2022-11-01","arxiv_id":"2211.00496","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-solve-voxel-building-embodied","slug":"learning-to-solve-voxel-building-embodied","title":"Learning to Solve Voxel Building Embodied Tasks from Pixels and Natural Language Instructions","date":"2022-11-01","arxiv_id":"2211.00688","repositories_listed":1,"syntology":null},{"url":"/paper/operator-selection-in-adaptive-large","slug":"operator-selection-in-adaptive-large","title":"Online Control of Adaptive Large Neighborhood Search using Deep Reinforcement Learning","date":"2022-11-01","arxiv_id":"2211.00759","repositories_listed":1,"syntology":null},{"url":"/paper/agent-time-attention-for-sparse-rewards-multi","slug":"agent-time-attention-for-sparse-rewards-multi","title":"Agent-Time Attention for Sparse Rewards Multi-Agent Reinforcement Learning","date":"2022-10-31","arxiv_id":"2210.17540","repositories_listed":1,"syntology":null},{"url":"/paper/disentangled-un-controllable-features","slug":"disentangled-un-controllable-features","title":"Disentangled (Un)Controllable Features","date":"2022-10-31","arxiv_id":"2211.00086","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-optimize-permutation-flow-shop","slug":"learning-to-optimize-permutation-flow-shop","title":"Learning to Optimize Permutation Flow Shop Scheduling via Graph-based Imitation Learning","date":"2022-10-31","arxiv_id":"2210.17178","repositories_listed":1,"syntology":null},{"url":"/paper/rlet-a-reinforcement-learning-based-approach","slug":"rlet-a-reinforcement-learning-based-approach","title":"RLET: A Reinforcement Learning Based Approach for Explainable QA with Entailment Trees","date":"2022-10-31","arxiv_id":"2210.17095","repositories_listed":1,"syntology":null},{"url":"/paper/bimrl-brain-inspired-meta-reinforcement","slug":"bimrl-brain-inspired-meta-reinforcement","title":"BIMRL: Brain Inspired Meta Reinforcement Learning","date":"2022-10-29","arxiv_id":"2210.16530","repositories_listed":1,"syntology":null},{"url":"/paper/goal-exploration-augmentation-via-pre-trained","slug":"goal-exploration-augmentation-via-pre-trained","title":"Goal Exploration Augmentation via Pre-trained Skills for Sparse-Reward Long-Horizon Goal-Conditioned Reinforcement Learning","date":"2022-10-28","arxiv_id":"2210.16058","repositories_listed":1,"syntology":null},{"url":"/paper/lad-language-augmented-diffusion-for","slug":"lad-language-augmented-diffusion-for","title":"Language Control Diffusion: Efficiently Scaling through Space, Time, and Tasks","date":"2022-10-27","arxiv_id":"2210.15629","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/lad-language-augmented-diffusion-for#ran","syntology_url":"https://syntology.ai/paper/2210.15629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.15629"}},"official":{"repos":["ezhang7423/language-control-diffusion"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/environment-design-for-inverse-reinforcement","slug":"environment-design-for-inverse-reinforcement","title":"Environment Design for Inverse Reinforcement Learning","date":"2022-10-26","arxiv_id":"2210.14972","repositories_listed":1,"syntology":null},{"url":"/paper/erl-re-2-efficient-evolutionary-reinforcement","slug":"erl-re-2-efficient-evolutionary-reinforcement","title":"ERL-Re$^2$: Efficient Evolutionary Reinforcement Learning with Shared State Representation and Individual Policy Representation","date":"2022-10-26","arxiv_id":"2210.17375","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/erl-re-2-efficient-evolutionary-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2210.17375","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.17375"}},"official":{"repos":["yeshenpy/erl-re2"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/low-rank-modular-reinforcement-learning-via","slug":"low-rank-modular-reinforcement-learning-via","title":"Low-Rank Modular Reinforcement Learning via Muscle Synergy","date":"2022-10-26","arxiv_id":"2210.15479","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/low-rank-modular-reinforcement-learning-via#ran","syntology_url":"https://syntology.ai/paper/2210.15479","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.15479"}},"official":{"repos":["drdh/synergy-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/provable-safe-reinforcement-learning-with","slug":"provable-safe-reinforcement-learning-with","title":"Provable Safe Reinforcement Learning with Binary Feedback","date":"2022-10-26","arxiv_id":"2210.14492","repositories_listed":1,"syntology":null},{"url":"/paper/aacher-assorted-actor-critic-deep","slug":"aacher-assorted-actor-critic-deep","title":"AACHER: Assorted Actor-Critic Deep Reinforcement Learning with Hindsight Experience Replay","date":"2022-10-24","arxiv_id":"2210.12892","repositories_listed":1,"syntology":null},{"url":"/paper/adlight-a-universal-approach-of-traffic","slug":"adlight-a-universal-approach-of-traffic","title":"ADLight: A Universal Approach of Traffic Signal Control with Augmented Data Using Reinforcement Learning","date":"2022-10-24","arxiv_id":"2210.13378","repositories_listed":1,"syntology":null},{"url":"/paper/energy-pricing-in-p2p-energy-systems-using","slug":"energy-pricing-in-p2p-energy-systems-using","title":"Energy Pricing in P2P Energy Systems Using Reinforcement Learning","date":"2022-10-24","arxiv_id":"2210.13555","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-long-term-memory-in-3d-mazes","slug":"evaluating-long-term-memory-in-3d-mazes","title":"Evaluating Long-Term Memory in 3D Mazes","date":"2022-10-24","arxiv_id":"2210.13383","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evaluating-long-term-memory-in-3d-mazes#ran","syntology_url":"https://syntology.ai/paper/2210.13383","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13383"}},"official":{"repos":["jurgisp/memory-maze"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/idrl-identifying-identities-in-multi-agent","slug":"idrl-identifying-identities-in-multi-agent","title":"Classifying Ambiguous Identities in Hidden-Role Stochastic Games with Multi-Agent Reinforcement Learning","date":"2022-10-24","arxiv_id":"2210.12896","repositories_listed":1,"syntology":null},{"url":"/paper/meet-a-monte-carlo-exploration-exploitation","slug":"meet-a-monte-carlo-exploration-exploitation","title":"MEET: A Monte Carlo Exploration-Exploitation Trade-off for Buffer Sampling","date":"2022-10-24","arxiv_id":"2210.13545","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-path-finding-via-tree-lstm","slug":"multi-agent-path-finding-via-tree-lstm","title":"Multi-Agent Path Finding via Tree LSTM","date":"2022-10-24","arxiv_id":"2210.12933","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-the-evolution-of-linear-regions","slug":"understanding-the-evolution-of-linear-regions","title":"Understanding the Evolution of Linear Regions in Deep Reinforcement Learning","date":"2022-10-24","arxiv_id":"2210.13611","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/understanding-the-evolution-of-linear-regions#ran","syntology_url":"https://syntology.ai/paper/2210.13611","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13611"}},"official":{"repos":["setarehc/deep_rl_regions"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/paco-parameter-compositional-multi-task","slug":"paco-parameter-compositional-multi-task","title":"PaCo: Parameter-Compositional Multi-Task Reinforcement Learning","date":"2022-10-21","arxiv_id":"2210.11653","repositories_listed":1,"syntology":null},{"url":"/paper/hypernetworks-in-meta-reinforcement-learning","slug":"hypernetworks-in-meta-reinforcement-learning","title":"Hypernetworks in Meta-Reinforcement Learning","date":"2022-10-20","arxiv_id":"2210.11348","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hypernetworks-in-meta-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2210.11348","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.11348"}},"official":{"repos":["jacooba/hyper"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rmbench-benchmarking-deep-reinforcement","slug":"rmbench-benchmarking-deep-reinforcement","title":"RMBench: Benchmarking Deep Reinforcement Learning for Robotic Manipulator Control","date":"2022-10-20","arxiv_id":"2210.11262","repositories_listed":1,"syntology":null},{"url":"/paper/the-pump-scheduling-problem-a-real-world","slug":"the-pump-scheduling-problem-a-real-world","title":"The Pump Scheduling Problem: A Real-World Scenario for Reinforcement Learning","date":"2022-10-20","arxiv_id":"2210.11111","repositories_listed":1,"syntology":null},{"url":"/paper/diambra-arena-a-new-reinforcement-learning","slug":"diambra-arena-a-new-reinforcement-learning","title":"DIAMBRA Arena: a New Reinforcement Learning Platform for Research and Experimentation","date":"2022-10-19","arxiv_id":"2210.10595","repositories_listed":1,"syntology":null},{"url":"/paper/learning-preferences-for-interactive-autonomy","slug":"learning-preferences-for-interactive-autonomy","title":"Learning Preferences for Interactive Autonomy","date":"2022-10-19","arxiv_id":"2210.10899","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-feasibility-of-cross-task-transfer","slug":"on-the-feasibility-of-cross-task-transfer","title":"On the Feasibility of Cross-Task Transfer with Model-Based Reinforcement Learning","date":"2022-10-19","arxiv_id":"2210.10763","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/on-the-feasibility-of-cross-task-transfer#ran","syntology_url":"https://syntology.ai/paper/2210.10763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10763"}},"official":{"repos":["mlpc-ucsd/xtra"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-offline-reinforcement-learning-with","slug":"robust-offline-reinforcement-learning-with","title":"Robust Offline Reinforcement Learning with Gradient Penalty and Constraint Relaxation","date":"2022-10-19","arxiv_id":"2210.10469","repositories_listed":1,"syntology":null},{"url":"/paper/when-to-ask-for-help-proactive-interventions","slug":"when-to-ask-for-help-proactive-interventions","title":"When to Ask for Help: Proactive Interventions in Autonomous Reinforcement Learning","date":"2022-10-19","arxiv_id":"2210.10765","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/when-to-ask-for-help-proactive-interventions#ran","syntology_url":"https://syntology.ai/paper/2210.10765","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10765"}},"official":{"repos":["tajwarfahim/proactive_interventions"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/ceip-combining-explicit-and-implicit-priors","slug":"ceip-combining-explicit-and-implicit-priors","title":"CEIP: Combining Explicit and Implicit Priors for Reinforcement Learning with Demonstrations","date":"2022-10-18","arxiv_id":"2210.09496","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/ceip-combining-explicit-and-implicit-priors#ran","syntology_url":"https://syntology.ai/paper/2210.09496","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.09496"}},"official":{"repos":["289371298/ceip"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/curriculum-reinforcement-learning-using","slug":"curriculum-reinforcement-learning-using","title":"Curriculum Reinforcement Learning using Optimal Transport via Gradual Domain Adaptation","date":"2022-10-18","arxiv_id":"2210.10195","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/curriculum-reinforcement-learning-using#ran","syntology_url":"https://syntology.ai/paper/2210.10195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10195"}},"official":{"repos":["peidehuang/gradient"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-black-box-reinforcement-learning-with","slug":"deep-black-box-reinforcement-learning-with","title":"Deep Black-Box Reinforcement Learning with Movement Primitives","date":"2022-10-18","arxiv_id":"2210.09622","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deep-black-box-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2210.09622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.09622"}},"official":{"repos":["ALRhub/fancy_gym"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-value-function-learning-for","slug":"rethinking-value-function-learning-for","title":"Rethinking Value Function Learning for Generalization in Reinforcement Learning","date":"2022-10-18","arxiv_id":"2210.09960","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/rethinking-value-function-learning-for#ran","syntology_url":"https://syntology.ai/paper/2210.09960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.09960"}},"official":{"repos":["snu-mllab/dcpg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-uncertainty-in-deep-state-space-models-for","slug":"on-uncertainty-in-deep-state-space-models-for","title":"On Uncertainty in Deep State Space Models for Model-Based Reinforcement Learning","date":"2022-10-17","arxiv_id":"2210.09256","repositories_listed":1,"syntology":null},{"url":"/paper/teacher-forcing-recovers-reward-functions-for","slug":"teacher-forcing-recovers-reward-functions-for","title":"Teacher Forcing Recovers Reward Functions for Text Generation","date":"2022-10-17","arxiv_id":"2210.08708","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/teacher-forcing-recovers-reward-functions-for#ran","syntology_url":"https://syntology.ai/paper/2210.08708","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.08708"}},"official":{"repos":["manga-uofa/lmreward"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-policy-guided-imitation-approach-for","slug":"a-policy-guided-imitation-approach-for","title":"A Policy-Guided Imitation Approach for Offline Reinforcement Learning","date":"2022-10-15","arxiv_id":"2210.08323","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/a-policy-guided-imitation-approach-for#ran","syntology_url":"https://syntology.ai/paper/2210.08323","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.08323"}},"official":{"repos":["ryanxhr/por"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/analyzing-the-robustness-of-pecnet","slug":"analyzing-the-robustness-of-pecnet","title":"G-PECNet: Towards a Generalizable Pedestrian Trajectory Prediction System","date":"2022-10-15","arxiv_id":"2210.09846","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/analyzing-the-robustness-of-pecnet#ran","syntology_url":"https://syntology.ai/paper/2210.09846","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.09846"}},"official":{"repos":["aryan-garg/pecnet-pedestrian-trajectory-prediction"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/when-to-update-your-model-constrained-model","slug":"when-to-update-your-model-constrained-model","title":"When to Update Your Model: Constrained Model-based Reinforcement Learning","date":"2022-10-15","arxiv_id":"2210.08349","repositories_listed":1,"syntology":null},{"url":"/paper/distributional-reward-estimation-for","slug":"distributional-reward-estimation-for","title":"Distributional Reward Estimation for Effective Multi-Agent Deep Reinforcement Learning","date":"2022-10-14","arxiv_id":"2210.07636","repositories_listed":1,"syntology":null},{"url":"/paper/just-round-quantized-observation-spaces","slug":"just-round-quantized-observation-spaces","title":"Just Round: Quantized Observation Spaces Enable Memory Efficient Learning of Dynamic Locomotion","date":"2022-10-14","arxiv_id":"2210.08065","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-safe-deep-reinforcement-learning","slug":"model-based-safe-deep-reinforcement-learning","title":"Model-based Safe Deep Reinforcement Learning via a Constrained Proximal Policy Optimization Algorithm","date":"2022-10-14","arxiv_id":"2210.07573","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/model-based-safe-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2210.07573","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07573"}},"official":{"repos":["akjayant/mbppol"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/mutual-information-regularized-offline-1","slug":"mutual-information-regularized-offline-1","title":"Mutual Information Regularized Offline Reinforcement Learning","date":"2022-10-14","arxiv_id":"2210.07484","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mutual-information-regularized-offline-1#ran","syntology_url":"https://syntology.ai/paper/2210.07484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07484"}},"official":{"repos":["sail-sg/misa"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-model-based-reinforcement-learning-with-2","slug":"safe-model-based-reinforcement-learning-with-2","title":"Safe Model-Based Reinforcement Learning with an Uncertainty-Aware Reachability Certificate","date":"2022-10-14","arxiv_id":"2210.07553","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safe-model-based-reinforcement-learning-with-2#ran","syntology_url":"https://syntology.ai/paper/2210.07553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07553"}},"official":{"repos":["ManUtdMoon/Safe_MBRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/skill-based-reinforcement-learning-with","slug":"skill-based-reinforcement-learning-with","title":"Skill-Based Reinforcement Learning with Intrinsic Reward Matching","date":"2022-10-14","arxiv_id":"2210.07426","repositories_listed":1,"syntology":null},{"url":"/paper/touplegdd-a-fine-designed-solution-of","slug":"touplegdd-a-fine-designed-solution-of","title":"ToupleGDD: A Fine-Designed Solution of Influence Maximization by Deep Reinforcement Learning","date":"2022-10-14","arxiv_id":"2210.07500","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/touplegdd-a-fine-designed-solution-of#ran","syntology_url":"https://syntology.ai/paper/2210.07500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07500"}},"official":{"repos":["Dtrycode/ToupleGDD"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-mixture-of-surprises-for-unsupervised","slug":"a-mixture-of-surprises-for-unsupervised","title":"A Mixture of Surprises for Unsupervised Reinforcement Learning","date":"2022-10-13","arxiv_id":"2210.06702","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/a-mixture-of-surprises-for-unsupervised#ran","syntology_url":"https://syntology.ai/paper/2210.06702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06702"}},"official":{"repos":["leaplabthu/moss"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/bootstrap-advantage-estimation-for-policy","slug":"bootstrap-advantage-estimation-for-policy","title":"Bootstrap Advantage Estimation for Policy Optimization in Reinforcement Learning","date":"2022-10-13","arxiv_id":"2210.07312","repositories_listed":1,"syntology":null},{"url":"/paper/harfang3d-dog-fight-sandbox-a-reinforcement","slug":"harfang3d-dog-fight-sandbox-a-reinforcement","title":"Harfang3D Dog-Fight Sandbox: A Reinforcement Learning Research Platform for the Customized Control Tasks of Fighter Aircrafts","date":"2022-10-13","arxiv_id":"2210.07282","repositories_listed":1,"syntology":null},{"url":"/paper/sustainable-online-reinforcement-learning-for","slug":"sustainable-online-reinforcement-learning-for","title":"Sustainable Online Reinforcement Learning for Auto-bidding","date":"2022-10-13","arxiv_id":"2210.07006","repositories_listed":1,"syntology":null},{"url":"/paper/visual-reinforcement-learning-with-self","slug":"visual-reinforcement-learning-with-self","title":"Visual Reinforcement Learning with Self-Supervised 3D Representations","date":"2022-10-13","arxiv_id":"2210.07241","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":14,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":2,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/visual-reinforcement-learning-with-self#ran","syntology_url":"https://syntology.ai/paper/2210.07241","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07241"}},"official":{"repos":["YanjieZe/rl3d"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-unified-framework-for-alternating-offline","slug":"a-unified-framework-for-alternating-offline","title":"A Unified Framework for Alternating Offline Model Training and Policy Learning","date":"2022-10-12","arxiv_id":"2210.05922","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-unified-framework-for-alternating-offline#ran","syntology_url":"https://syntology.ai/paper/2210.05922","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05922"}},"official":{"repos":["shentao-yang/ampl_neurips2022"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/centralized-training-with-hybrid-execution-in","slug":"centralized-training-with-hybrid-execution-in","title":"Centralized Training with Hybrid Execution in Multi-Agent Reinforcement Learning","date":"2022-10-12","arxiv_id":"2210.06274","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-adversarial-training-without","slug":"efficient-adversarial-training-without","title":"Efficient Adversarial Training without Attacking: Worst-Case-Aware Robust Reinforcement Learning","date":"2022-10-12","arxiv_id":"2210.05927","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-adversarial-training-without#ran","syntology_url":"https://syntology.ai/paper/2210.05927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05927"}},"official":{"repos":["umd-huang-lab/wocar-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/semi-supervised-offline-reinforcement-1","slug":"semi-supervised-offline-reinforcement-1","title":"Semi-Supervised Offline Reinforcement Learning with Action-Free Trajectories","date":"2022-10-12","arxiv_id":"2210.06518","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/semi-supervised-offline-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2210.06518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06518"}},"official":{"repos":["facebookresearch/ssorl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/conserweightive-behavioral-cloning-for","slug":"conserweightive-behavioral-cloning-for","title":"Reliable Conditioning of Behavioral Cloning for Offline Reinforcement Learning","date":"2022-10-11","arxiv_id":"2210.05158","repositories_listed":1,"syntology":null},{"url":"/paper/dhrl-a-graph-based-approach-for-long-horizon","slug":"dhrl-a-graph-based-approach-for-long-horizon","title":"DHRL: A Graph-Based Approach for Long-Horizon and Sparse Hierarchical Reinforcement Learning","date":"2022-10-11","arxiv_id":"2210.05150","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dhrl-a-graph-based-approach-for-long-horizon#ran","syntology_url":"https://syntology.ai/paper/2210.05150","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05150"}},"official":null}},{"url":"/paper/leco-learnable-episodic-count-for-task","slug":"leco-learnable-episodic-count-for-task","title":"LECO: Learnable Episodic Count for Task-Specific Intrinsic Reward","date":"2022-10-11","arxiv_id":"2210.05409","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":1,"n_ran_checked":1,"n_instrument":6,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":10,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/leco-learnable-episodic-count-for-task#ran","syntology_url":"https://syntology.ai/paper/2210.05409","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05409"}},"official":{"repos":["kakaobrain/leco"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/marllib-extending-rllib-for-multi-agent","slug":"marllib-extending-rllib-for-multi-agent","title":"MARLlib: A Scalable and Efficient Multi-agent Reinforcement Learning Library","date":"2022-10-11","arxiv_id":"2210.13708","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/marllib-extending-rllib-for-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2210.13708","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13708"}},"official":{"repos":["replicable-marl/marllib"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mastering-the-game-of-no-press-diplomacy-via","slug":"mastering-the-game-of-no-press-diplomacy-via","title":"Mastering the Game of No-Press Diplomacy via Human-Regularized Reinforcement Learning and Planning","date":"2022-10-11","arxiv_id":"2210.05492","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mastering-the-game-of-no-press-diplomacy-via#ran","syntology_url":"https://syntology.ai/paper/2210.05492","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05492"}},"official":null}},{"url":"/paper/a-comprehensive-survey-of-data-augmentation","slug":"a-comprehensive-survey-of-data-augmentation","title":"A Comprehensive Survey of Data Augmentation in Visual Reinforcement Learning","date":"2022-10-10","arxiv_id":"2210.04561","repositories_listed":1,"syntology":null},{"url":"/paper/a-policy-gradient-approach-for-finite-horizon","slug":"a-policy-gradient-approach-for-finite-horizon","title":"A policy gradient approach for Finite Horizon Constrained Markov Decision Processes","date":"2022-10-10","arxiv_id":"2210.04527","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/a-policy-gradient-approach-for-finite-horizon#ran","syntology_url":"https://syntology.ai/paper/2210.04527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.04527"}},"official":{"repos":["gsoumyajit/Finite-Horizon-with-constraints"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/benchmarking-reinforcement-learning-1","slug":"benchmarking-reinforcement-learning-1","title":"Benchmarking Reinforcement Learning Techniques for Autonomous Navigation","date":"2022-10-10","arxiv_id":"2210.04839","repositories_listed":1,"syntology":{"n":6,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/benchmarking-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2210.04839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.04839"}},"official":null}},{"url":"/paper/experiential-explanations-for-reinforcement","slug":"experiential-explanations-for-reinforcement","title":"Experiential Explanations for Reinforcement Learning","date":"2022-10-10","arxiv_id":"2210.04723","repositories_listed":1,"syntology":null},{"url":"/paper/learning-credit-assignment-for-cooperative","slug":"learning-credit-assignment-for-cooperative","title":"Learning Explicit Credit Assignment for Cooperative Multi-Agent Reinforcement Learning via Polarization Policy Gradient","date":"2022-10-10","arxiv_id":"2210.05367","repositories_listed":1,"syntology":null},{"url":"/paper/multiagent-reinforcement-learning-based-on","slug":"multiagent-reinforcement-learning-based-on","title":"Multiagent Reinforcement Learning Based on Fusion-Multiactor-Attention-Critic for Multiple-Unmanned-Aerial-Vehicle Navigation Control","date":"2022-10-10","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/decomposed-mutual-information-optimization","slug":"decomposed-mutual-information-optimization","title":"Decomposed Mutual Information Optimization for Generalized Context in Meta-Reinforcement Learning","date":"2022-10-09","arxiv_id":"2210.04209","repositories_listed":1,"syntology":null},{"url":"/paper/flexible-attention-based-multi-policy-fusion-1","slug":"flexible-attention-based-multi-policy-fusion-1","title":"Flexible Attention-Based Multi-Policy Fusion for Efficient Deep Reinforcement Learning","date":"2022-10-07","arxiv_id":"2210.03729","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/flexible-attention-based-multi-policy-fusion-1#ran","syntology_url":"https://syntology.ai/paper/2210.03729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03729"}},"official":{"repos":["pascalson/kgrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mind-your-data-hiding-backdoors-in-offline","slug":"mind-your-data-hiding-backdoors-in-offline","title":"BAFFLE: Hiding Backdoors in Offline Reinforcement Learning Datasets","date":"2022-10-07","arxiv_id":"2210.04688","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-directed-controller-synthesis-via","slug":"scaling-directed-controller-synthesis-via","title":"Exploration Policies for On-the-Fly Controller Synthesis: A Reinforcement Learning Approach","date":"2022-10-07","arxiv_id":"2210.05393","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based-evasion","slug":"deep-reinforcement-learning-based-evasion","title":"Deep Reinforcement Learning based Evasion Generative Adversarial Network for Botnet Detection","date":"2022-10-06","arxiv_id":"2210.02840","repositories_listed":1,"syntology":null},{"url":"/paper/neuroevolution-is-a-competitive-alternative","slug":"neuroevolution-is-a-competitive-alternative","title":"Neuroevolution is a Competitive Alternative to Reinforcement Learning for Skill Discovery","date":"2022-10-06","arxiv_id":"2210.03516","repositories_listed":1,"syntology":null}],"record_sha256":"08858715eb0f7ff9dbfb94e6da67bc2e039b0402dc9bd72b4570b179dbbb41b2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}