{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/23","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":23,"pages_in_order":152,"rows_per_page":100,"rows":[2201,2300],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/22","next":"/task/reinforcement-learning-1/papers/24","papers":[{"url":"/paper/convlab-3-a-flexible-dialogue-system-toolkit","slug":"convlab-3-a-flexible-dialogue-system-toolkit","title":"ConvLab-3: A Flexible Dialogue System Toolkit Based on a Unified Data Format","date":"2022-11-30","arxiv_id":"2211.17148","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-reinforcement-learning-through","slug":"efficient-reinforcement-learning-through","title":"Efficient Reinforcement Learning Through Trajectory Generation","date":"2022-11-30","arxiv_id":"2211.17249","repositories_listed":1,"syntology":null},{"url":"/paper/general-policy-mapping-online-continual","slug":"general-policy-mapping-online-continual","title":"General policy mapping: online continual reinforcement learning inspired on the insect brain","date":"2022-11-30","arxiv_id":"2211.16759","repositories_listed":1,"syntology":null},{"url":"/paper/one-risk-to-rule-them-all-a-risk-sensitive-1","slug":"one-risk-to-rule-them-all-a-risk-sensitive-1","title":"One Risk to Rule Them All: A Risk-Sensitive Perspective on Model-Based Offline Reinforcement Learning","date":"2022-11-30","arxiv_id":"2212.00124","repositories_listed":1,"syntology":{"n":23,"n_ran":14,"n_constructed":4,"n_ran_checked":8,"n_instrument":6,"n_unverified":9,"n_honours":2,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"14 ran (of which 4 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 0 violated, 6 with no contract checked; 6 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/one-risk-to-rule-them-all-a-risk-sensitive-1#ran","syntology_url":"https://syntology.ai/paper/2212.00124","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.00124"}},"official":{"repos":["marc-rigter/1r2r"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["found_in_text","official","unlocated"]}}},{"url":"/paper/real-time-bidding-strategy-in-display","slug":"real-time-bidding-strategy-in-display","title":"Real-time Bidding Strategy in Display Advertising: An Empirical Analysis","date":"2022-11-30","arxiv_id":"2212.02222","repositories_listed":1,"syntology":null},{"url":"/paper/reinforced-language-modeling-for-end-to-end","slug":"reinforced-language-modeling-for-end-to-end","title":"KRLS: Improving End-to-End Response Generation in Task Oriented Dialog with Reinforced Keywords Learning","date":"2022-11-30","arxiv_id":"2211.16773","repositories_listed":1,"syntology":null},{"url":"/paper/welfare-and-fairness-in-multi-objective","slug":"welfare-and-fairness-in-multi-objective","title":"Welfare and Fairness in Multi-objective Reinforcement Learning","date":"2022-11-30","arxiv_id":"2212.01382","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/welfare-and-fairness-in-multi-objective#ran","syntology_url":"https://syntology.ai/paper/2212.01382","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.01382"}},"official":{"repos":["MuhangTian/Fair-MORL-AAMAS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/behavior-estimation-from-multi-source-data","slug":"behavior-estimation-from-multi-source-data","title":"Behavior Estimation from Multi-Source Data for Offline Reinforcement Learning","date":"2022-11-29","arxiv_id":"2211.16078","repositories_listed":1,"syntology":null},{"url":"/paper/improved-representation-of-asymmetrical","slug":"improved-representation-of-asymmetrical","title":"Improved Representation of Asymmetrical Distances with Interval Quasimetric Embeddings","date":"2022-11-28","arxiv_id":"2211.15120","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improved-representation-of-asymmetrical#ran","syntology_url":"https://syntology.ai/paper/2211.15120","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.15120"}},"official":{"repos":["quasimetric-learning/torch-quasimetric"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/quantile-constrained-reinforcement-learning-a","slug":"quantile-constrained-reinforcement-learning-a","title":"Quantile Constrained Reinforcement Learning: A Reinforcement Learning Framework Constraining Outage Probability","date":"2022-11-28","arxiv_id":"2211.15034","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/quantile-constrained-reinforcement-learning-a#ran","syntology_url":"https://syntology.ai/paper/2211.15034","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.15034"}},"official":{"repos":["wyjung0625/qcpo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/applying-deep-reinforcement-learning-to-the","slug":"applying-deep-reinforcement-learning-to-the","title":"Applying Deep Reinforcement Learning to the HP Model for Protein Structure Prediction","date":"2022-11-27","arxiv_id":"2211.14939","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/applying-deep-reinforcement-learning-to-the#ran","syntology_url":"https://syntology.ai/paper/2211.14939","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.14939"}},"official":{"repos":["compsoftmatterbiophysics-cityu-hk/applying-drl-to-hp-model-for-protein-structure-prediction"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/bear-physics-principled-building-environment","slug":"bear-physics-principled-building-environment","title":"BEAR: Physics-Principled Building Environment for Control and Reinforcement Learning","date":"2022-11-27","arxiv_id":"2211.14744","repositories_listed":1,"syntology":null},{"url":"/paper/assistive-teaching-of-motor-control-tasks-to","slug":"assistive-teaching-of-motor-control-tasks-to","title":"Assistive Teaching of Motor Control Tasks to Humans","date":"2022-11-25","arxiv_id":"2211.14003","repositories_listed":1,"syntology":null},{"url":"/paper/explainable-and-safe-reinforcement-learning","slug":"explainable-and-safe-reinforcement-learning","title":"Explainable and Safe Reinforcement Learning for Autonomous Air Mobility","date":"2022-11-24","arxiv_id":"2211.13474","repositories_listed":1,"syntology":null},{"url":"/paper/actively-learning-costly-reward-functions-for","slug":"actively-learning-costly-reward-functions-for","title":"Actively Learning Costly Reward Functions for Reinforcement Learning","date":"2022-11-23","arxiv_id":"2211.13260","repositories_listed":1,"syntology":null},{"url":"/paper/masked-autoencoding-for-scalable-and","slug":"masked-autoencoding-for-scalable-and","title":"Masked Autoencoding for Scalable and Generalizable Decision Making","date":"2022-11-23","arxiv_id":"2211.12740","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/masked-autoencoding-for-scalable-and#ran","syntology_url":"https://syntology.ai/paper/2211.12740","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.12740"}},"official":{"repos":["fangchenliu/maskdp_public"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-reinforcement-learning-badminton","slug":"a-reinforcement-learning-badminton","title":"A Reinforcement Learning Badminton Environment for Simulating Player Tactics (Student Abstract)","date":"2022-11-22","arxiv_id":"2211.12234","repositories_listed":1,"syntology":null},{"url":"/paper/monte-carlo-forest-search-unsat-solver","slug":"monte-carlo-forest-search-unsat-solver","title":"UNSAT Solver Synthesis via Monte Carlo Forest Search","date":"2022-11-22","arxiv_id":"2211.12581","repositories_listed":1,"syntology":null},{"url":"/paper/a-low-latency-adaptive-coding-spiking","slug":"a-low-latency-adaptive-coding-spiking","title":"A Low Latency Adaptive Coding Spiking Framework for Deep Reinforcement Learning","date":"2022-11-21","arxiv_id":"2211.11760","repositories_listed":1,"syntology":null},{"url":"/paper/examining-policy-entropy-of-reinforcement","slug":"examining-policy-entropy-of-reinforcement","title":"Examining Policy Entropy of Reinforcement Learning Agents for Personalization Tasks","date":"2022-11-21","arxiv_id":"2211.11869","repositories_listed":1,"syntology":null},{"url":"/paper/tempera-test-time-prompting-via-reinforcement","slug":"tempera-test-time-prompting-via-reinforcement","title":"TEMPERA: Test-Time Prompting via Reinforcement Learning","date":"2022-11-21","arxiv_id":"2211.11890","repositories_listed":1,"syntology":null},{"url":"/paper/tinyqmix-distributed-access-control-for-mmtc","slug":"tinyqmix-distributed-access-control-for-mmtc","title":"TinyQMIX: Distributed Access Control for mMTC via Multi-agent Reinforcement Learning","date":"2022-11-21","arxiv_id":"2211.11692","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-cheap-talk","slug":"adversarial-cheap-talk","title":"Adversarial Cheap Talk","date":"2022-11-20","arxiv_id":"2211.11030","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-meta-reinforcement-learning-for","slug":"efficient-meta-reinforcement-learning-for","title":"Efficient Meta Reinforcement Learning for Preference-based Fast Adaptation","date":"2022-11-20","arxiv_id":"2211.10861","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-meta-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2211.10861","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.10861"}},"official":{"repos":["stilwell-git/adaptation-with-noisy-oracle"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-search-for-job-shop-scheduling","slug":"learning-to-search-for-job-shop-scheduling","title":"Deep Reinforcement Learning Guided Improvement Heuristic for Job Shop Scheduling","date":"2022-11-20","arxiv_id":"2211.10936","repositories_listed":1,"syntology":null},{"url":"/paper/safelight-a-reinforcement-learning-method","slug":"safelight-a-reinforcement-learning-method","title":"SafeLight: A Reinforcement Learning Method toward Collision-free Traffic Signal Control","date":"2022-11-20","arxiv_id":"2211.10871","repositories_listed":1,"syntology":null},{"url":"/paper/debiasing-meta-gradient-reinforcement","slug":"debiasing-meta-gradient-reinforcement","title":"Debiasing Meta-Gradient Reinforcement Learning by Learning the Outer Value Function","date":"2022-11-19","arxiv_id":"2211.10550","repositories_listed":1,"syntology":null},{"url":"/paper/reinform-selecting-paths-with-reinforcement","slug":"reinform-selecting-paths-with-reinforcement","title":"ReInform: Selecting paths with reinforcement learning for contextualized link prediction","date":"2022-11-19","arxiv_id":"2211.10688","repositories_listed":1,"syntology":null},{"url":"/paper/gosum-extractive-summarization-of-long","slug":"gosum-extractive-summarization-of-long","title":"GoSum: Extractive Summarization of Long Documents by Reinforcement Learning and Graph Organized discourse state","date":"2022-11-18","arxiv_id":"2211.10247","repositories_listed":1,"syntology":null},{"url":"/paper/language-conditioned-reinforcement-learning","slug":"language-conditioned-reinforcement-learning","title":"Language-Conditioned Reinforcement Learning to Solve Misunderstandings with Action Corrections","date":"2022-11-18","arxiv_id":"2211.10168","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-conditioned-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2211.10168","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.10168"}},"official":{"repos":["frankroeder/lanro-gym"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/provable-defense-against-backdoor-policies-in","slug":"provable-defense-against-backdoor-policies-in","title":"Provable Defense against Backdoor Policies in Reinforcement Learning","date":"2022-11-18","arxiv_id":"2211.10530","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/provable-defense-against-backdoor-policies-in#ran","syntology_url":"https://syntology.ai/paper/2211.10530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.10530"}},"official":{"repos":["skbharti/provable-defense-in-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/agent-state-construction-with-auxiliary","slug":"agent-state-construction-with-auxiliary","title":"Agent-State Construction with Auxiliary Inputs","date":"2022-11-15","arxiv_id":"2211.07805","repositories_listed":1,"syntology":null},{"url":"/paper/explainable-action-advising-for-multi-agent","slug":"explainable-action-advising-for-multi-agent","title":"Explainable Action Advising for Multi-Agent Reinforcement Learning","date":"2022-11-15","arxiv_id":"2211.07882","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchically-structured-task-agnostic","slug":"hierarchically-structured-task-agnostic","title":"Hierarchically Structured Task-Agnostic Continual Learning","date":"2022-11-14","arxiv_id":"2211.07725","repositories_listed":1,"syntology":null},{"url":"/paper/interactively-learning-to-summarise-timelines","slug":"interactively-learning-to-summarise-timelines","title":"Towards Abstractive Timeline Summarisation using Preference-based Reinforcement Learning","date":"2022-11-14","arxiv_id":"2211.07596","repositories_listed":1,"syntology":null},{"url":"/paper/redeeming-intrinsic-rewards-via-constrained","slug":"redeeming-intrinsic-rewards-via-constrained","title":"Redeeming Intrinsic Rewards via Constrained Optimization","date":"2022-11-14","arxiv_id":"2211.07627","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/redeeming-intrinsic-rewards-via-constrained#ran","syntology_url":"https://syntology.ai/paper/2211.07627","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.07627"}},"official":{"repos":["improbable-ai/eipo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-data-driven-offline-simulations-for","slug":"towards-data-driven-offline-simulations-for","title":"Towards Data-Driven Offline Simulations for Online Reinforcement Learning","date":"2022-11-14","arxiv_id":"2211.07614","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-data-driven-offline-simulations-for#ran","syntology_url":"https://syntology.ai/paper/2211.07614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.07614"}},"official":{"repos":["microsoft/rl-offline-simulation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-heterogeneous-agent-cooperation-via","slug":"learning-heterogeneous-agent-cooperation-via","title":"Learning Heterogeneous Agent Cooperation via Multiagent League Training","date":"2022-11-13","arxiv_id":"2211.11616","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-explainable-reinforcement","slug":"a-survey-on-explainable-reinforcement","title":"A Survey on Explainable Reinforcement Learning: Concepts, Algorithms, Challenges","date":"2022-11-12","arxiv_id":"2211.06665","repositories_listed":1,"syntology":null},{"url":"/paper/online-anomalous-subtrajectory-detection-on","slug":"online-anomalous-subtrajectory-detection-on","title":"Online Anomalous Subtrajectory Detection on Road Networks with Deep Reinforcement Learning","date":"2022-11-12","arxiv_id":"2211.08415","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-deep-reinforcement-learning-with-1","slug":"efficient-deep-reinforcement-learning-with-1","title":"Efficient Deep Reinforcement Learning with Predictive Processing Proximal Policy Optimization","date":"2022-11-11","arxiv_id":"2211.06236","repositories_listed":1,"syntology":null},{"url":"/paper/global-and-local-analysis-of-interestingness","slug":"global-and-local-analysis-of-interestingness","title":"Global and Local Analysis of Interestingness for Competency-Aware Deep Reinforcement Learning","date":"2022-11-11","arxiv_id":"2211.06376","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-in-an-adaptable-chess","slug":"reinforcement-learning-in-an-adaptable-chess","title":"Reinforcement Learning in an Adaptable Chess Environment for Detecting Human-understandable Concepts","date":"2022-11-10","arxiv_id":"2211.05500","repositories_listed":1,"syntology":null},{"url":"/paper/deep-w-networks-solving-multi-objective","slug":"deep-w-networks-solving-multi-objective","title":"Deep W-Networks: Solving Multi-Objective Optimisation Problems With Deep Reinforcement Learning","date":"2022-11-09","arxiv_id":"2211.04813","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-sequentiality-in-reinforcement","slug":"leveraging-sequentiality-in-reinforcement","title":"Leveraging Sequentiality in Reinforcement Learning from a Single Demonstration","date":"2022-11-09","arxiv_id":"2211.04786","repositories_listed":1,"syntology":null},{"url":"/paper/doubly-inhomogeneous-reinforcement-learning","slug":"doubly-inhomogeneous-reinforcement-learning","title":"Doubly Inhomogeneous Reinforcement Learning","date":"2022-11-08","arxiv_id":"2211.03983","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-follow-instructions-in-text-based","slug":"learning-to-follow-instructions-in-text-based","title":"Learning to Follow Instructions in Text-Based Games","date":"2022-11-08","arxiv_id":"2211.04591","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-follow-instructions-in-text-based#ran","syntology_url":"https://syntology.ai/paper/2211.04591","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.04591"}},"official":{"repos":["mathieutuli/ltl-gata"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/curriculum-based-asymmetric-multi-task","slug":"curriculum-based-asymmetric-multi-task","title":"Curriculum-based Asymmetric Multi-task Reinforcement Learning","date":"2022-11-07","arxiv_id":"2211.03352","repositories_listed":1,"syntology":null},{"url":"/paper/design-process-is-a-reinforcement-learning","slug":"design-process-is-a-reinforcement-learning","title":"Design Process is a Reinforcement Learning Problem","date":"2022-11-06","arxiv_id":"2211.03136","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-quality-diversity-algorithms-on","slug":"benchmarking-quality-diversity-algorithms-on","title":"Benchmarking Quality-Diversity Algorithms on Neuroevolution for Reinforcement Learning","date":"2022-11-04","arxiv_id":"2211.02193","repositories_listed":1,"syntology":null},{"url":"/paper/de-novo-protac-design-using-graph-based-deep","slug":"de-novo-protac-design-using-graph-based-deep","title":"De novo PROTAC design using graph-based deep generative models","date":"2022-11-04","arxiv_id":"2211.02660","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/de-novo-protac-design-using-graph-based-deep#ran","syntology_url":"https://syntology.ai/paper/2211.02660","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.02660"}},"official":{"repos":["divnori/protac-design"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diversity-based-deep-reinforcement-learning","slug":"diversity-based-deep-reinforcement-learning","title":"Diversity-based Deep Reinforcement Learning Towards Multidimensional Difficulty for Fighting Game AI","date":"2022-11-04","arxiv_id":"2211.02759","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diversity-based-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2211.02759","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.02759"}},"official":{"repos":["emily-halina/brisket"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/residual-skill-policies-learning-an-adaptable","slug":"residual-skill-policies-learning-an-adaptable","title":"Residual Skill Policies: Learning an Adaptable Skill-based Action Space for Reinforcement Learning for Robotics","date":"2022-11-04","arxiv_id":"2211.02231","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/residual-skill-policies-learning-an-adaptable#ran","syntology_url":"https://syntology.ai/paper/2211.02231","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.02231"}},"official":{"repos":["krishanrana/reskill"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-benefits-of-model-based-generalization-in","slug":"the-benefits-of-model-based-generalization-in","title":"The Benefits of Model-Based Generalization in Reinforcement Learning","date":"2022-11-04","arxiv_id":"2211.02222","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-benefits-of-model-based-generalization-in#ran","syntology_url":"https://syntology.ai/paper/2211.02222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.02222"}},"official":{"repos":["kenjyoung/model_generalization_code_supplement"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-safety-in-model-based-reinforcement","slug":"learning-safety-in-model-based-reinforcement","title":"Learning safety in model-based Reinforcement Learning using MPC and Gaussian Processes","date":"2022-11-03","arxiv_id":"2211.01860","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-fully-observable-policies-for","slug":"leveraging-fully-observable-policies-for","title":"Leveraging Fully Observable Policies for Learning under Partial Observability","date":"2022-11-03","arxiv_id":"2211.01991","repositories_listed":1,"syntology":null},{"url":"/paper/synthesis-of-separation-processes-with","slug":"synthesis-of-separation-processes-with","title":"Synthesis of separation processes with reinforcement learning","date":"2022-11-03","arxiv_id":"2211.04327","repositories_listed":1,"syntology":null},{"url":"/paper/behavior-prior-representation-learning-for","slug":"behavior-prior-representation-learning-for","title":"Behavior Prior Representation learning for Offline Reinforcement Learning","date":"2022-11-02","arxiv_id":"2211.00863","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/behavior-prior-representation-learning-for#ran","syntology_url":"https://syntology.ai/paper/2211.00863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.00863"}},"official":{"repos":["bit1029public/offline_bpr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamiclight-dynamically-tuning-traffic","slug":"dynamiclight-dynamically-tuning-traffic","title":"DynamicLight: Two-Stage Dynamic Traffic Signal Timing","date":"2022-11-02","arxiv_id":"2211.01025","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-for-13","slug":"multi-agent-reinforcement-learning-for-13","title":"Multi-Agent Reinforcement Learning for Adaptive Mesh Refinement","date":"2022-11-02","arxiv_id":"2211.00801","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/multi-agent-reinforcement-learning-for-13#ran","syntology_url":"https://syntology.ai/paper/2211.00801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.00801"}},"official":{"repos":["011235813/marl-amr"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/spatial-temporal-recurrent-reinforcement","slug":"spatial-temporal-recurrent-reinforcement","title":"Spatial-temporal recurrent reinforcement learning for autonomous ships","date":"2022-11-02","arxiv_id":"2211.01004","repositories_listed":1,"syntology":null},{"url":"/paper/can-maker-taker-fees-prevent-algorithmic","slug":"can-maker-taker-fees-prevent-algorithmic","title":"Can maker-taker fees prevent algorithmic cooperation in market making?","date":"2022-11-01","arxiv_id":"2211.00496","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-solve-voxel-building-embodied","slug":"learning-to-solve-voxel-building-embodied","title":"Learning to Solve Voxel Building Embodied Tasks from Pixels and Natural Language Instructions","date":"2022-11-01","arxiv_id":"2211.00688","repositories_listed":1,"syntology":null},{"url":"/paper/operator-selection-in-adaptive-large","slug":"operator-selection-in-adaptive-large","title":"Online Control of Adaptive Large Neighborhood Search using Deep Reinforcement Learning","date":"2022-11-01","arxiv_id":"2211.00759","repositories_listed":1,"syntology":null},{"url":"/paper/agent-time-attention-for-sparse-rewards-multi","slug":"agent-time-attention-for-sparse-rewards-multi","title":"Agent-Time Attention for Sparse Rewards Multi-Agent Reinforcement Learning","date":"2022-10-31","arxiv_id":"2210.17540","repositories_listed":1,"syntology":null},{"url":"/paper/disentangled-un-controllable-features","slug":"disentangled-un-controllable-features","title":"Disentangled (Un)Controllable Features","date":"2022-10-31","arxiv_id":"2211.00086","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-optimize-permutation-flow-shop","slug":"learning-to-optimize-permutation-flow-shop","title":"Learning to Optimize Permutation Flow Shop Scheduling via Graph-based Imitation Learning","date":"2022-10-31","arxiv_id":"2210.17178","repositories_listed":1,"syntology":null},{"url":"/paper/rlet-a-reinforcement-learning-based-approach","slug":"rlet-a-reinforcement-learning-based-approach","title":"RLET: A Reinforcement Learning Based Approach for Explainable QA with Entailment Trees","date":"2022-10-31","arxiv_id":"2210.17095","repositories_listed":1,"syntology":null},{"url":"/paper/bimrl-brain-inspired-meta-reinforcement","slug":"bimrl-brain-inspired-meta-reinforcement","title":"BIMRL: Brain Inspired Meta Reinforcement Learning","date":"2022-10-29","arxiv_id":"2210.16530","repositories_listed":1,"syntology":null},{"url":"/paper/goal-exploration-augmentation-via-pre-trained","slug":"goal-exploration-augmentation-via-pre-trained","title":"Goal Exploration Augmentation via Pre-trained Skills for Sparse-Reward Long-Horizon Goal-Conditioned Reinforcement Learning","date":"2022-10-28","arxiv_id":"2210.16058","repositories_listed":1,"syntology":null},{"url":"/paper/lad-language-augmented-diffusion-for","slug":"lad-language-augmented-diffusion-for","title":"Language Control Diffusion: Efficiently Scaling through Space, Time, and Tasks","date":"2022-10-27","arxiv_id":"2210.15629","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/lad-language-augmented-diffusion-for#ran","syntology_url":"https://syntology.ai/paper/2210.15629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.15629"}},"official":{"repos":["ezhang7423/language-control-diffusion"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/environment-design-for-inverse-reinforcement","slug":"environment-design-for-inverse-reinforcement","title":"Environment Design for Inverse Reinforcement Learning","date":"2022-10-26","arxiv_id":"2210.14972","repositories_listed":1,"syntology":null},{"url":"/paper/erl-re-2-efficient-evolutionary-reinforcement","slug":"erl-re-2-efficient-evolutionary-reinforcement","title":"ERL-Re$^2$: Efficient Evolutionary Reinforcement Learning with Shared State Representation and Individual Policy Representation","date":"2022-10-26","arxiv_id":"2210.17375","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/erl-re-2-efficient-evolutionary-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2210.17375","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.17375"}},"official":{"repos":["yeshenpy/erl-re2"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/low-rank-modular-reinforcement-learning-via","slug":"low-rank-modular-reinforcement-learning-via","title":"Low-Rank Modular Reinforcement Learning via Muscle Synergy","date":"2022-10-26","arxiv_id":"2210.15479","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/low-rank-modular-reinforcement-learning-via#ran","syntology_url":"https://syntology.ai/paper/2210.15479","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.15479"}},"official":{"repos":["drdh/synergy-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/provable-safe-reinforcement-learning-with","slug":"provable-safe-reinforcement-learning-with","title":"Provable Safe Reinforcement Learning with Binary Feedback","date":"2022-10-26","arxiv_id":"2210.14492","repositories_listed":1,"syntology":null},{"url":"/paper/shortest-edit-path-crossover-a-theory-driven","slug":"shortest-edit-path-crossover-a-theory-driven","title":"Shortest Edit Path Crossover: A Theory-driven Solution to the Permutation Problem in Evolutionary Neural Architecture Search","date":"2022-10-25","arxiv_id":"2210.14016","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shortest-edit-path-crossover-a-theory-driven#ran","syntology_url":"https://syntology.ai/paper/2210.14016","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.14016"}},"official":{"repos":["cognizant-ai-labs/sepx-paper"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/sim-to-real-via-sim-to-seg-end-to-end-off","slug":"sim-to-real-via-sim-to-seg-end-to-end-off","title":"Sim-to-Real via Sim-to-Seg: End-to-end Off-road Autonomous Driving Without Real Data","date":"2022-10-25","arxiv_id":"2210.14721","repositories_listed":1,"syntology":null},{"url":"/paper/teal-learning-accelerated-optimization-of","slug":"teal-learning-accelerated-optimization-of","title":"Teal: Learning-Accelerated Optimization of WAN Traffic Engineering","date":"2022-10-25","arxiv_id":"2210.13763","repositories_listed":1,"syntology":null},{"url":"/paper/aacher-assorted-actor-critic-deep","slug":"aacher-assorted-actor-critic-deep","title":"AACHER: Assorted Actor-Critic Deep Reinforcement Learning with Hindsight Experience Replay","date":"2022-10-24","arxiv_id":"2210.12892","repositories_listed":1,"syntology":null},{"url":"/paper/adlight-a-universal-approach-of-traffic","slug":"adlight-a-universal-approach-of-traffic","title":"ADLight: A Universal Approach of Traffic Signal Control with Augmented Data Using Reinforcement Learning","date":"2022-10-24","arxiv_id":"2210.13378","repositories_listed":1,"syntology":null},{"url":"/paper/avalon-a-benchmark-for-rl-generalization","slug":"avalon-a-benchmark-for-rl-generalization","title":"Avalon: A Benchmark for RL Generalization Using Procedurally Generated Worlds","date":"2022-10-24","arxiv_id":"2210.13417","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/avalon-a-benchmark-for-rl-generalization#ran","syntology_url":"https://syntology.ai/paper/2210.13417","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13417"}},"official":null}},{"url":"/paper/dichotomy-of-control-separating-what-you-can","slug":"dichotomy-of-control-separating-what-you-can","title":"Dichotomy of Control: Separating What You Can Control from What You Cannot","date":"2022-10-24","arxiv_id":"2210.13435","repositories_listed":1,"syntology":null},{"url":"/paper/energy-pricing-in-p2p-energy-systems-using","slug":"energy-pricing-in-p2p-energy-systems-using","title":"Energy Pricing in P2P Energy Systems Using Reinforcement Learning","date":"2022-10-24","arxiv_id":"2210.13555","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-long-term-memory-in-3d-mazes","slug":"evaluating-long-term-memory-in-3d-mazes","title":"Evaluating Long-Term Memory in 3D Mazes","date":"2022-10-24","arxiv_id":"2210.13383","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evaluating-long-term-memory-in-3d-mazes#ran","syntology_url":"https://syntology.ai/paper/2210.13383","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13383"}},"official":{"repos":["jurgisp/memory-maze"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/idrl-identifying-identities-in-multi-agent","slug":"idrl-identifying-identities-in-multi-agent","title":"Classifying Ambiguous Identities in Hidden-Role Stochastic Games with Multi-Agent Reinforcement Learning","date":"2022-10-24","arxiv_id":"2210.12896","repositories_listed":1,"syntology":null},{"url":"/paper/meet-a-monte-carlo-exploration-exploitation","slug":"meet-a-monte-carlo-exploration-exploitation","title":"MEET: A Monte Carlo Exploration-Exploitation Trade-off for Buffer Sampling","date":"2022-10-24","arxiv_id":"2210.13545","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-path-finding-via-tree-lstm","slug":"multi-agent-path-finding-via-tree-lstm","title":"Multi-Agent Path Finding via Tree LSTM","date":"2022-10-24","arxiv_id":"2210.12933","repositories_listed":1,"syntology":null},{"url":"/paper/symbolic-distillation-for-learned-tcp","slug":"symbolic-distillation-for-learned-tcp","title":"Symbolic Distillation for Learned TCP Congestion Control","date":"2022-10-24","arxiv_id":"2210.16987","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/symbolic-distillation-for-learned-tcp#ran","syntology_url":"https://syntology.ai/paper/2210.16987","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.16987"}},"official":{"repos":["vita-group/symbolicpcc"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-the-evolution-of-linear-regions","slug":"understanding-the-evolution-of-linear-regions","title":"Understanding the Evolution of Linear Regions in Deep Reinforcement Learning","date":"2022-10-24","arxiv_id":"2210.13611","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/understanding-the-evolution-of-linear-regions#ran","syntology_url":"https://syntology.ai/paper/2210.13611","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13611"}},"official":{"repos":["setarehc/deep_rl_regions"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/biologically-plausible-variational-policy","slug":"biologically-plausible-variational-policy","title":"Biologically Plausible Variational Policy Gradient with Spiking Recurrent Winner-Take-All Networks","date":"2022-10-21","arxiv_id":"2210.13225","repositories_listed":1,"syntology":null},{"url":"/paper/paco-parameter-compositional-multi-task","slug":"paco-parameter-compositional-multi-task","title":"PaCo: Parameter-Compositional Multi-Task Reinforcement Learning","date":"2022-10-21","arxiv_id":"2210.11653","repositories_listed":1,"syntology":null},{"url":"/paper/rate-splitting-for-intelligent-reflecting","slug":"rate-splitting-for-intelligent-reflecting","title":"Rate-Splitting for Intelligent Reflecting Surface-Aided Multiuser VR Streaming","date":"2022-10-21","arxiv_id":"2210.12191","repositories_listed":1,"syntology":null},{"url":"/paper/hypernetworks-in-meta-reinforcement-learning","slug":"hypernetworks-in-meta-reinforcement-learning","title":"Hypernetworks in Meta-Reinforcement Learning","date":"2022-10-20","arxiv_id":"2210.11348","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hypernetworks-in-meta-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2210.11348","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.11348"}},"official":{"repos":["jacooba/hyper"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mocoda-model-based-counterfactual-data","slug":"mocoda-model-based-counterfactual-data","title":"MoCoDA: Model-based Counterfactual Data Augmentation","date":"2022-10-20","arxiv_id":"2210.11287","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mocoda-model-based-counterfactual-data#ran","syntology_url":"https://syntology.ai/paper/2210.11287","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.11287"}},"official":{"repos":["spitis/mocoda"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/rmbench-benchmarking-deep-reinforcement","slug":"rmbench-benchmarking-deep-reinforcement","title":"RMBench: Benchmarking Deep Reinforcement Learning for Robotic Manipulator Control","date":"2022-10-20","arxiv_id":"2210.11262","repositories_listed":1,"syntology":null},{"url":"/paper/task-phasing-automated-curriculum-learning","slug":"task-phasing-automated-curriculum-learning","title":"Task Phasing: Automated Curriculum Learning from Demonstrations","date":"2022-10-20","arxiv_id":"2210.10999","repositories_listed":1,"syntology":null},{"url":"/paper/the-pump-scheduling-problem-a-real-world","slug":"the-pump-scheduling-problem-a-real-world","title":"The Pump Scheduling Problem: A Real-World Scenario for Reinforcement Learning","date":"2022-10-20","arxiv_id":"2210.11111","repositories_listed":1,"syntology":null},{"url":"/paper/clutr-curriculum-learning-via-unsupervised","slug":"clutr-curriculum-learning-via-unsupervised","title":"CLUTR: Curriculum Learning via Unsupervised Task Representation Learning","date":"2022-10-19","arxiv_id":"2210.10243","repositories_listed":1,"syntology":null},{"url":"/paper/diambra-arena-a-new-reinforcement-learning","slug":"diambra-arena-a-new-reinforcement-learning","title":"DIAMBRA Arena: a New Reinforcement Learning Platform for Research and Experimentation","date":"2022-10-19","arxiv_id":"2210.10595","repositories_listed":1,"syntology":null},{"url":"/paper/learning-preferences-for-interactive-autonomy","slug":"learning-preferences-for-interactive-autonomy","title":"Learning Preferences for Interactive Autonomy","date":"2022-10-19","arxiv_id":"2210.10899","repositories_listed":1,"syntology":null}],"record_sha256":"5c477ca162ef0d6fef1c79097f8e86b23504e93e34f85f246858f756096e4077","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}