{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/27","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":27,"pages_in_order":152,"rows_per_page":100,"rows":[2601,2700],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/26","next":"/task/reinforcement-learning-1/papers/28","papers":[{"url":"/paper/learning-to-branch-with-tree-mdps","slug":"learning-to-branch-with-tree-mdps","title":"Learning to branch with Tree MDPs","date":"2022-05-23","arxiv_id":"2205.11107","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-branch-with-tree-mdps#ran","syntology_url":"https://syntology.ai/paper/2205.11107","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.11107"}},"official":{"repos":["lascavana/rl2branch"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/using-natural-language-and-program","slug":"using-natural-language-and-program","title":"Using Natural Language and Program Abstractions to Instill Human Inductive Biases in Machines","date":"2022-05-23","arxiv_id":"2205.11558","repositories_listed":1,"syntology":null},{"url":"/paper/memory-efficient-reinforcement-learning-with","slug":"memory-efficient-reinforcement-learning-with","title":"Memory-efficient Reinforcement Learning with Value-based Knowledge Consolidation","date":"2022-05-22","arxiv_id":"2205.10868","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/memory-efficient-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2205.10868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.10868"}},"official":{"repos":["qlan3/MeDQN"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/user-interactive-offline-reinforcement","slug":"user-interactive-offline-reinforcement","title":"User-Interactive Offline Reinforcement Learning","date":"2022-05-21","arxiv_id":"2205.10629","repositories_listed":1,"syntology":null},{"url":"/paper/a-review-of-safe-reinforcement-learning","slug":"a-review-of-safe-reinforcement-learning","title":"A Review of Safe Reinforcement Learning: Methods, Theory and Applications","date":"2022-05-20","arxiv_id":"2205.10330","repositories_listed":1,"syntology":null},{"url":"/paper/arlo-a-framework-for-automated-reinforcement","slug":"arlo-a-framework-for-automated-reinforcement","title":"ARLO: A Framework for Automated Reinforcement Learning","date":"2022-05-20","arxiv_id":"2205.10416","repositories_listed":1,"syntology":null},{"url":"/paper/learning-task-relevant-representations-for","slug":"learning-task-relevant-representations-for","title":"Learning Task-relevant Representations for Generalization via Characteristic Functions of Reward Sequence Distributions","date":"2022-05-20","arxiv_id":"2205.10218","repositories_listed":1,"syntology":null},{"url":"/paper/self-paced-multi-agent-reinforcement-learning","slug":"self-paced-multi-agent-reinforcement-learning","title":"Learning Progress Driven Multi-Agent Curriculum","date":"2022-05-20","arxiv_id":"2205.10016","repositories_listed":1,"syntology":null},{"url":"/paper/synthesis-from-satisficing-and-temporal-goals","slug":"synthesis-from-satisficing-and-temporal-goals","title":"Synthesis from Satisficing and Temporal Goals","date":"2022-05-20","arxiv_id":"2205.10464","repositories_listed":1,"syntology":null},{"url":"/paper/towards-biologically-plausible-dreaming-and","slug":"towards-biologically-plausible-dreaming-and","title":"Towards biologically plausible Dreaming and Planning in recurrent spiking networks","date":"2022-05-20","arxiv_id":"2205.10044","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-greedy-search-tracking-by-multi-agent","slug":"beyond-greedy-search-tracking-by-multi-agent","title":"Beyond Greedy Search: Tracking by Multi-Agent Reinforcement Learning-based Beam Search","date":"2022-05-19","arxiv_id":"2205.09676","repositories_listed":1,"syntology":null},{"url":"/paper/deconfounding-actor-critic-network-with","slug":"deconfounding-actor-critic-network-with","title":"Deconfounding Actor-Critic Network with Policy Adaptation for Dynamic Treatment Regimes","date":"2022-05-19","arxiv_id":"2205.09852","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-time-1","slug":"deep-reinforcement-learning-for-time-1","title":"Deep Reinforcement Learning for Time Allocation and Directional Transmission in Joint Radar-Communication","date":"2022-05-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-brain-inspired","slug":"reinforcement-learning-with-brain-inspired","title":"Reinforcement Learning with Brain-Inspired Modulation can Improve Adaptation to Environmental Changes","date":"2022-05-19","arxiv_id":"2205.09729","repositories_listed":1,"syntology":null},{"url":"/paper/time-series-anomaly-detection-via","slug":"time-series-anomaly-detection-via","title":"Time Series Anomaly Detection via Reinforcement Learning-Based Model Selection","date":"2022-05-19","arxiv_id":"2205.09884","repositories_listed":1,"syntology":null},{"url":"/paper/a2c-is-a-special-case-of-ppo","slug":"a2c-is-a-special-case-of-ppo","title":"A2C is a special case of PPO","date":"2022-05-18","arxiv_id":"2205.09123","repositories_listed":1,"syntology":null},{"url":"/paper/neighborhood-mixup-experience-replay-local","slug":"neighborhood-mixup-experience-replay-local","title":"Neighborhood Mixup Experience Replay: Local Convex Interpolation for Improved Sample Efficiency in Continuous Control Tasks","date":"2022-05-18","arxiv_id":"2205.09117","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-adaptive-prediction-intervals-for","slug":"optimal-adaptive-prediction-intervals-for","title":"Optimal Adaptive Prediction Intervals for Electricity Load Forecasting in Distribution Systems via Reinforcement Learning","date":"2022-05-18","arxiv_id":"2205.08698","repositories_listed":1,"syntology":null},{"url":"/paper/deepsim-a-reinforcement-learning-environment","slug":"deepsim-a-reinforcement-learning-environment","title":"DeepSim: A Reinforcement Learning Environment Build Toolkit for ROS and Gazebo","date":"2022-05-17","arxiv_id":"2205.08034","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-unsupervised-sentence-compression-1","slug":"efficient-unsupervised-sentence-compression-1","title":"Efficient Unsupervised Sentence Compression by Fine-tuning Transformers with Reinforcement Learning","date":"2022-05-17","arxiv_id":"2205.08221","repositories_listed":1,"syntology":null},{"url":"/paper/the-primacy-bias-in-deep-reinforcement","slug":"the-primacy-bias-in-deep-reinforcement","title":"The Primacy Bias in Deep Reinforcement Learning","date":"2022-05-16","arxiv_id":"2205.07802","repositories_listed":1,"syntology":null},{"url":"/paper/unified-distributed-environment","slug":"unified-distributed-environment","title":"Unified Distributed Environment","date":"2022-05-14","arxiv_id":"2205.06946","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-computational","slug":"deep-reinforcement-learning-for-computational","title":"Deep Reinforcement Learning for Computational Fluid Dynamics on HPC Systems","date":"2022-05-13","arxiv_id":"2205.06502","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-transmission-control-for-wireless","slug":"distributed-transmission-control-for-wireless","title":"Distributed Transmission Control for Wireless Networks using Multi-Agent Reinforcement Learning","date":"2022-05-13","arxiv_id":"2205.06800","repositories_listed":1,"syntology":null},{"url":"/paper/modularity-in-neat-reinforcement-learning","slug":"modularity-in-neat-reinforcement-learning","title":"Towards Understanding the Link Between Modularity and Performance in Neural Networks for Reinforcement Learning","date":"2022-05-13","arxiv_id":"2205.06451","repositories_listed":1,"syntology":null},{"url":"/paper/upside-down-reinforcement-learning-can","slug":"upside-down-reinforcement-learning-can","title":"Upside-Down Reinforcement Learning Can Diverge in Stochastic Environments With Episodic Resets","date":"2022-05-13","arxiv_id":"2205.06595","repositories_listed":1,"syntology":null},{"url":"/paper/a-state-distribution-matching-approach-to-non","slug":"a-state-distribution-matching-approach-to-non","title":"A State-Distribution Matching Approach to Non-Episodic Reinforcement Learning","date":"2022-05-11","arxiv_id":"2205.05212","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-state-distribution-matching-approach-to-non#ran","syntology_url":"https://syntology.ai/paper/2205.05212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.05212"}},"official":{"repos":["architsharma97/medal"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/intelligent-reflecting-surface-configurations","slug":"intelligent-reflecting-surface-configurations","title":"Intelligent Reflecting Surface Configurations for Smart Radio Using Deep Reinforcement Learning","date":"2022-05-11","arxiv_id":"2205.05269","repositories_listed":1,"syntology":null},{"url":"/paper/gamma-and-vega-hedging-using-deep","slug":"gamma-and-vega-hedging-using-deep","title":"Gamma and Vega Hedging Using Deep Distributional Reinforcement Learning","date":"2022-05-10","arxiv_id":"2205.05614","repositories_listed":1,"syntology":null},{"url":"/paper/state-encoders-in-reinforcement-learning-for","slug":"state-encoders-in-reinforcement-learning-for","title":"State Encoders in Reinforcement Learning for Recommendation: A Reproducibility Study","date":"2022-05-10","arxiv_id":"2205.04797","repositories_listed":1,"syntology":null},{"url":"/paper/vesnet-rl-simulation-based-reinforcement","slug":"vesnet-rl-simulation-based-reinforcement","title":"VesNet-RL: Simulation-based Reinforcement Learning for Real-World US Probe Navigation","date":"2022-05-10","arxiv_id":"2205.06676","repositories_listed":1,"syntology":null},{"url":"/paper/dxformer-a-decoupled-automatic-diagnostic","slug":"dxformer-a-decoupled-automatic-diagnostic","title":"DxFormer: A Decoupled Automatic Diagnostic System Based on Decoder-Encoder Transformer with Dense Symptom Representations","date":"2022-05-08","arxiv_id":"2205.03755","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-brachiate-via-simplified-model","slug":"learning-to-brachiate-via-simplified-model","title":"Learning to Brachiate via Simplified Model Imitation","date":"2022-05-08","arxiv_id":"2205.03943","repositories_listed":1,"syntology":null},{"url":"/paper/simultaneous-double-q-learning-with","slug":"simultaneous-double-q-learning-with","title":"Simultaneous Double Q-learning with Conservative Advantage Learning for Actor-Critic Methods","date":"2022-05-08","arxiv_id":"2205.03819","repositories_listed":1,"syntology":null},{"url":"/paper/rlflow-optimising-neural-network-subgraph","slug":"rlflow-optimising-neural-network-subgraph","title":"RLFlow: Optimising Neural Network Subgraph Transformation with World Models","date":"2022-05-03","arxiv_id":"2205.01435","repositories_listed":1,"syntology":null},{"url":"/paper/cclf-a-contrastive-curiosity-driven-learning","slug":"cclf-a-contrastive-curiosity-driven-learning","title":"CCLF: A Contrastive-Curiosity-Driven Learning Framework for Sample-Efficient Reinforcement Learning","date":"2022-05-02","arxiv_id":"2205.00943","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/cclf-a-contrastive-curiosity-driven-learning#ran","syntology_url":"https://syntology.ai/paper/2205.00943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.00943"}},"official":{"repos":["csun001/cclf"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/large-neighborhood-search-based-on-neural","slug":"large-neighborhood-search-based-on-neural","title":"Large Neighborhood Search based on Neural Construction Heuristics","date":"2022-05-02","arxiv_id":"2205.00772","repositories_listed":1,"syntology":null},{"url":"/paper/reinforced-cross-modal-alignment-for","slug":"reinforced-cross-modal-alignment-for","title":"Reinforced Cross-modal Alignment for Radiology Report Generation","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/ttopt-a-maximum-volume-quantized-tensor-train","slug":"ttopt-a-maximum-volume-quantized-tensor-train","title":"TTOpt: A Maximum Volume Quantized Tensor Train-based Optimization and its Application to Reinforcement Learning","date":"2022-04-30","arxiv_id":"2205.00293","repositories_listed":1,"syntology":{"n":17,"n_ran":14,"n_constructed":0,"n_ran_checked":6,"n_instrument":8,"n_unverified":3,"n_honours":6,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 6 honoured, 0 violated, 0 with no contract checked; 8 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ttopt-a-maximum-volume-quantized-tensor-train#ran","syntology_url":"https://syntology.ai/paper/2205.00293","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.00293"}},"official":{"repos":["andreichertkov/ttopt"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/cost-effective-mlaas-federation-a","slug":"cost-effective-mlaas-federation-a","title":"Cost Effective MLaaS Federation: A Combinatorial Reinforcement Learning Approach","date":"2022-04-29","arxiv_id":"2204.13971","repositories_listed":1,"syntology":null},{"url":"/paper/markov-abstractions-for-pac-reinforcement","slug":"markov-abstractions-for-pac-reinforcement","title":"Markov Abstractions for PAC Reinforcement Learning in Non-Markov Decision Processes","date":"2022-04-29","arxiv_id":"2205.01053","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/markov-abstractions-for-pac-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2205.01053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.01053"}},"official":{"repos":["whitemech/markov-abstractions-code-ijcai22"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/accelerating-robot-learning-of-contact-rich","slug":"accelerating-robot-learning-of-contact-rich","title":"Accelerating Robot Learning of Contact-Rich Manipulations: A Curriculum Learning Study","date":"2022-04-27","arxiv_id":"2204.12844","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-for-11","slug":"multi-agent-reinforcement-learning-for-11","title":"Multi-Agent Reinforcement Learning for Traffic Signal Control through Universal Communication Method","date":"2022-04-26","arxiv_id":"2204.12190","repositories_listed":1,"syntology":null},{"url":"/paper/social-learning-spontaneously-emerges-by","slug":"social-learning-spontaneously-emerges-by","title":"Social learning spontaneously emerges by searching optimal heuristics with deep reinforcement learning","date":"2022-04-26","arxiv_id":"2204.12371","repositories_listed":1,"syntology":null},{"url":"/paper/toward-policy-explanations-for-multi-agent","slug":"toward-policy-explanations-for-multi-agent","title":"Toward Policy Explanations for Multi-Agent Reinforcement Learning","date":"2022-04-26","arxiv_id":"2204.12568","repositories_listed":1,"syntology":null},{"url":"/paper/hypernca-growing-developmental-networks-with","slug":"hypernca-growing-developmental-networks-with","title":"HyperNCA: Growing Developmental Networks with Neural Cellular Automata","date":"2022-04-25","arxiv_id":"2204.11674","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hypernca-growing-developmental-networks-with#ran","syntology_url":"https://syntology.ai/paper/2204.11674","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.11674"}},"official":null}},{"url":"/paper/multi-objective-pointer-network-for","slug":"multi-objective-pointer-network-for","title":"Multi-objective Pointer Network for Combinatorial Optimization","date":"2022-04-25","arxiv_id":"2204.11860","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-objective-pointer-network-for#ran","syntology_url":"https://syntology.ai/paper/2204.11860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.11860"}},"official":{"repos":["gaoly/mopn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/predicting-real-time-scientific-experiments","slug":"predicting-real-time-scientific-experiments","title":"Predicting Real-time Scientific Experiments Using Transformer models and Reinforcement Learning","date":"2022-04-25","arxiv_id":"2204.11718","repositories_listed":1,"syntology":null},{"url":"/paper/towards-evaluating-adaptivity-of-model-based","slug":"towards-evaluating-adaptivity-of-model-based","title":"Towards Evaluating Adaptivity of Model-Based Reinforcement Learning Methods","date":"2022-04-25","arxiv_id":"2204.11464","repositories_listed":1,"syntology":null},{"url":"/paper/reward-reports-for-reinforcement-learning","slug":"reward-reports-for-reinforcement-learning","title":"Reward Reports for Reinforcement Learning","date":"2022-04-22","arxiv_id":"2204.10817","repositories_listed":1,"syntology":null},{"url":"/paper/6gan-ipv6-multi-pattern-target-generation-via","slug":"6gan-ipv6-multi-pattern-target-generation-via","title":"6GAN: IPv6 Multi-Pattern Target Generation via Generative Adversarial Nets with Reinforcement Learning","date":"2022-04-21","arxiv_id":"2204.09839","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-gaussian-mixture-critic-in-off","slug":"revisiting-gaussian-mixture-critic-in-off","title":"Revisiting Gaussian mixture critics in off-policy reinforcement learning: a sample-based approach","date":"2022-04-21","arxiv_id":"2204.10256","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforcement-learning-based-volt-var","slug":"a-reinforcement-learning-based-volt-var","title":"A Reinforcement Learning-based Volt-VAR Control Dataset and Testing Environment","date":"2022-04-20","arxiv_id":"2204.09500","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-a-two-echelon","slug":"deep-reinforcement-learning-for-a-two-echelon","title":"Comparing Deep Reinforcement Learning Algorithms in Two-Echelon Supply Chains","date":"2022-04-20","arxiv_id":"2204.09603","repositories_listed":1,"syntology":null},{"url":"/paper/coptidice-offline-constrained-reinforcement-1","slug":"coptidice-offline-constrained-reinforcement-1","title":"COptiDICE: Offline Constrained Reinforcement Learning via Stationary Distribution Correction Estimation","date":"2022-04-19","arxiv_id":"2204.08957","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/coptidice-offline-constrained-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2204.08957","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.08957"}},"official":{"repos":["deepmind/constrained_optidice"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/fedkl-tackling-data-heterogeneity-in","slug":"fedkl-tackling-data-heterogeneity-in","title":"FedKL: Tackling Data Heterogeneity in Federated Reinforcement Learning by Penalizing KL Divergence","date":"2022-04-18","arxiv_id":"2204.08125","repositories_listed":1,"syntology":null},{"url":"/paper/resource-constrained-neural-architecture","slug":"resource-constrained-neural-architecture","title":"TabNAS: Rejection Sampling for Neural Architecture Search on Tabular Datasets","date":"2022-04-15","arxiv_id":"2204.07615","repositories_listed":1,"syntology":null},{"url":"/paper/safe-reinforcement-learning-using-black-box","slug":"safe-reinforcement-learning-using-black-box","title":"Safe Reinforcement Learning Using Black-Box Reachability Analysis","date":"2022-04-15","arxiv_id":"2204.07417","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-game-playing-agents-with","slug":"understanding-game-playing-agents-with","title":"Understanding Game-Playing Agents with Natural Language Annotations","date":"2022-04-15","arxiv_id":"2204.07531","repositories_listed":1,"syntology":null},{"url":"/paper/can-question-rewriting-help-conversational","slug":"can-question-rewriting-help-conversational","title":"Can Question Rewriting Help Conversational Question Answering?","date":"2022-04-13","arxiv_id":"2204.06239","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-question-rewriting-help-conversational#ran","syntology_url":"https://syntology.ai/paper/2204.06239","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.06239"}},"official":{"repos":["hltchkust/cqr4cqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/gtlo-a-generalized-and-non-linear-multi","slug":"gtlo-a-generalized-and-non-linear-multi","title":"gTLO: A Generalized and Non-linear Multi-Objective Deep Reinforcement Learning Approach","date":"2022-04-11","arxiv_id":"2204.04988","repositories_listed":1,"syntology":null},{"url":"/paper/jorldy-a-fully-customizable-open-source","slug":"jorldy-a-fully-customizable-open-source","title":"JORLDY: a fully customizable open source framework for reinforcement learning","date":"2022-04-11","arxiv_id":"2204.04892","repositories_listed":1,"syntology":null},{"url":"/paper/confidence-estimation-transformer-for-long","slug":"confidence-estimation-transformer-for-long","title":"Confidence Estimation Transformer for Long-term Renewable Energy Forecasting in Reinforcement Learning-based Power Grid Dispatching","date":"2022-04-10","arxiv_id":"2204.04612","repositories_listed":1,"syntology":null},{"url":"/paper/driving-black-box-quantum-thermal-machines","slug":"driving-black-box-quantum-thermal-machines","title":"Model-free optimization of power/efficiency tradeoffs in quantum thermal machines using reinforcement learning","date":"2022-04-10","arxiv_id":"2204.04785","repositories_listed":1,"syntology":null},{"url":"/paper/offline-reinforcement-learning-for-safer","slug":"offline-reinforcement-learning-for-safer","title":"Offline Reinforcement Learning for Safer Blood Glucose Control in People with Type 1 Diabetes","date":"2022-04-07","arxiv_id":"2204.03376","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-alignment-for-history-representation","slug":"temporal-alignment-for-history-representation","title":"Temporal Alignment for History Representation in Reinforcement Learning","date":"2022-04-07","arxiv_id":"2204.03525","repositories_listed":1,"syntology":null},{"url":"/paper/federated-reinforcement-learning-with","slug":"federated-reinforcement-learning-with","title":"Federated Reinforcement Learning with Environment Heterogeneity","date":"2022-04-06","arxiv_id":"2204.02634","repositories_listed":1,"syntology":null},{"url":"/paper/automating-reinforcement-learning-with","slug":"automating-reinforcement-learning-with","title":"Automating Reinforcement Learning with Example-based Resets","date":"2022-04-05","arxiv_id":"2204.02041","repositories_listed":1,"syntology":null},{"url":"/paper/inferring-rewards-from-language-in-context","slug":"inferring-rewards-from-language-in-context","title":"Inferring Rewards from Language in Context","date":"2022-04-05","arxiv_id":"2204.02515","repositories_listed":1,"syntology":null},{"url":"/paper/jump-start-reinforcement-learning","slug":"jump-start-reinforcement-learning","title":"Jump-Start Reinforcement Learning","date":"2022-04-05","arxiv_id":"2204.02372","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-bid-long-term-multi-agent","slug":"learning-to-bid-long-term-multi-agent","title":"Learning to Bid Long-Term: Multi-Agent Reinforcement Learning with Long-Term and Sparse Reward in Repeated Auction Games","date":"2022-04-05","arxiv_id":"2204.02268","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-distributed-reinforcement","slug":"multi-agent-distributed-reinforcement","title":"Multi-Agent Distributed Reinforcement Learning for Making Decentralized Offloading Decisions","date":"2022-04-05","arxiv_id":"2204.02267","repositories_listed":1,"syntology":null},{"url":"/paper/disentangling-abstraction-from-statistical","slug":"disentangling-abstraction-from-statistical","title":"Disentangling Abstraction from Statistical Pattern Matching in Human and Machine Learning","date":"2022-04-04","arxiv_id":"2204.01437","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-agents-in-colonel","slug":"reinforcement-learning-agents-in-colonel","title":"Reinforcement Learning Agents in Colonel Blotto","date":"2022-04-04","arxiv_id":"2204.02785","repositories_listed":1,"syntology":null},{"url":"/paper/value-gradient-weighted-model-based-1","slug":"value-gradient-weighted-model-based-1","title":"Value Gradient weighted Model-Based Reinforcement Learning","date":"2022-04-04","arxiv_id":"2204.01464","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforcement-learning-approach-to-sensing","slug":"a-reinforcement-learning-approach-to-sensing","title":"A Reinforcement Learning Approach to Sensing Design in Resource-Constrained Wireless Networked Control Systems","date":"2022-04-01","arxiv_id":"2204.00703","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-guided-by-provable","slug":"reinforcement-learning-guided-by-provable","title":"Reinforcement Learning Guided by Provable Normative Compliance","date":"2022-03-30","arxiv_id":"2203.16275","repositories_listed":1,"syntology":null},{"url":"/paper/text-driven-video-acceleration-a-weakly","slug":"text-driven-video-acceleration-a-weakly","title":"Text-Driven Video Acceleration: A Weakly-Supervised Reinforcement Learning Method","date":"2022-03-29","arxiv_id":"2203.15778","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-risk-tendency-nano-drone-navigation","slug":"adaptive-risk-tendency-nano-drone-navigation","title":"Adaptive Risk-Tendency: Nano Drone Navigation in Cluttered Environments with Distributional Reinforcement Learning","date":"2022-03-28","arxiv_id":"2203.14749","repositories_listed":1,"syntology":null},{"url":"/paper/image-quality-assessment-for-machine-learning","slug":"image-quality-assessment-for-machine-learning","title":"Image quality assessment for machine learning tasks using meta-reinforcement learning","date":"2022-03-27","arxiv_id":"2203.14258","repositories_listed":1,"syntology":null},{"url":"/paper/an-optical-controlling-environment-and","slug":"an-optical-controlling-environment-and","title":"An Optical Control Environment for Benchmarking Reinforcement Learning Algorithms","date":"2022-03-23","arxiv_id":"2203.12114","repositories_listed":1,"syntology":null},{"url":"/paper/asynchronous-reinforcement-learning-for-real","slug":"asynchronous-reinforcement-learning-for-real","title":"Asynchronous Reinforcement Learning for Real-Time Control of Physical Robots","date":"2022-03-23","arxiv_id":"2203.12759","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/asynchronous-reinforcement-learning-for-real#ran","syntology_url":"https://syntology.ai/paper/2203.12759","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.12759"}},"official":{"repos":["yufengyuan/ur5_async_rl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/possibility-before-utility-learning-and-using-1","slug":"possibility-before-utility-learning-and-using-1","title":"Possibility Before Utility: Learning And Using Hierarchical Affordances","date":"2022-03-23","arxiv_id":"2203.12686","repositories_listed":1,"syntology":null},{"url":"/paper/insights-from-the-neurips-2021-nethack","slug":"insights-from-the-neurips-2021-nethack","title":"Insights From the NeurIPS 2021 NetHack Challenge","date":"2022-03-22","arxiv_id":"2203.11889","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/insights-from-the-neurips-2021-nethack#ran","syntology_url":"https://syntology.ai/paper/2203.11889","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.11889"}},"official":null}},{"url":"/paper/is-vanilla-policy-gradient-overlooked","slug":"is-vanilla-policy-gradient-overlooked","title":"Is Vanilla Policy Gradient Overlooked? Analyzing Deep Reinforcement Learning for Hanabi","date":"2022-03-22","arxiv_id":"2203.11656","repositories_listed":1,"syntology":null},{"url":"/paper/long-short-term-memory-for-spatial-encoding","slug":"long-short-term-memory-for-spatial-encoding","title":"Long Short-Term Memory for Spatial Encoding in Multi-Agent Path Planning","date":"2022-03-21","arxiv_id":"2203.10823","repositories_listed":1,"syntology":null},{"url":"/paper/reccover-detecting-causal-confusion-for","slug":"reccover-detecting-causal-confusion-for","title":"ReCCoVER: Detecting Causal Confusion for Explainable Reinforcement Learning","date":"2022-03-21","arxiv_id":"2203.11211","repositories_listed":1,"syntology":null},{"url":"/paper/microracer-a-didactic-environment-for-deep","slug":"microracer-a-didactic-environment-for-deep","title":"MicroRacer: a didactic environment for Deep Reinforcement Learning","date":"2022-03-20","arxiv_id":"2203.10494","repositories_listed":1,"syntology":null},{"url":"/paper/perceiving-the-world-question-guided","slug":"perceiving-the-world-question-guided","title":"Perceiving the World: Question-guided Reinforcement Learning for Text-based Games","date":"2022-03-20","arxiv_id":"2204.09597","repositories_listed":1,"syntology":null},{"url":"/paper/quantum-multi-agent-reinforcement-learning","slug":"quantum-multi-agent-reinforcement-learning","title":"Quantum Multi-Agent Reinforcement Learning via Variational Quantum Circuit Design","date":"2022-03-20","arxiv_id":"2203.10443","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-automatic","slug":"reinforcement-learning-for-automatic","title":"Reinforcement learning for automatic quadrilateral mesh generation: a soft actor-critic approach","date":"2022-03-19","arxiv_id":"2203.11203","repositories_listed":1,"syntology":null},{"url":"/paper/teachable-reinforcement-learning-via-advice-1","slug":"teachable-reinforcement-learning-via-advice-1","title":"Teachable Reinforcement Learning via Advice Distillation","date":"2022-03-19","arxiv_id":"2203.11197","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/teachable-reinforcement-learning-via-advice-1#ran","syntology_url":"https://syntology.ai/paper/2203.11197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.11197"}},"official":{"repos":["rll-research/teachable"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/gac-a-deep-reinforcement-learning-model","slug":"gac-a-deep-reinforcement-learning-model","title":"GAC: A Deep Reinforcement Learning Model Toward User Incentivization in Unknown Social Networks","date":"2022-03-17","arxiv_id":"2203.09578","repositories_listed":1,"syntology":null},{"url":"/paper/semi-markov-offline-reinforcement-learning","slug":"semi-markov-offline-reinforcement-learning","title":"Semi-Markov Offline Reinforcement Learning for Healthcare","date":"2022-03-17","arxiv_id":"2203.09365","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/semi-markov-offline-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2203.09365","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.09365"}},"official":{"repos":["mary-wu/smdp"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/coach-assisted-multi-agent-reinforcement","slug":"coach-assisted-multi-agent-reinforcement","title":"Coach-assisted Multi-Agent Reinforcement Learning Framework for Unexpected Crashed Agents","date":"2022-03-16","arxiv_id":"2203.08454","repositories_listed":1,"syntology":null},{"url":"/paper/copa-certifying-robust-policies-for-offline-1","slug":"copa-certifying-robust-policies-for-offline-1","title":"COPA: Certifying Robust Policies for Offline Reinforcement Learning against Poisoning Attacks","date":"2022-03-16","arxiv_id":"2203.08398","repositories_listed":1,"syntology":null},{"url":"/paper/ctds-centralized-teacher-with-decentralized","slug":"ctds-centralized-teacher-with-decentralized","title":"CTDS: Centralized Teacher with Decentralized Student for Multi-Agent Reinforcement Learning","date":"2022-03-16","arxiv_id":"2203.08412","repositories_listed":1,"syntology":null},{"url":"/paper/latent-variable-advantage-weighted-policy","slug":"latent-variable-advantage-weighted-policy","title":"Latent-Variable Advantage-Weighted Policy Optimization for Offline RL","date":"2022-03-16","arxiv_id":"2203.08949","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/latent-variable-advantage-weighted-policy#ran","syntology_url":"https://syntology.ai/paper/2203.08949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08949"}},"official":null}},{"url":"/paper/pmic-improving-multi-agent-reinforcement-1","slug":"pmic-improving-multi-agent-reinforcement-1","title":"PMIC: Improving Multi-Agent Reinforcement Learning with Progressive Mutual Information Collaboration","date":"2022-03-16","arxiv_id":"2203.08553","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pmic-improving-multi-agent-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2203.08553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08553"}},"official":{"repos":["yeshenpy/pmic"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/zipfian-environments-for-reinforcement","slug":"zipfian-environments-for-reinforcement","title":"Zipfian environments for Reinforcement Learning","date":"2022-03-15","arxiv_id":"2203.08222","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/zipfian-environments-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2203.08222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08222"}},"official":{"repos":["deepmind/zipfian_environments"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"2e0547b1bdedde3fea0139d7335564a8c9f365556ae6b8d840cb314deb57d306","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}