{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/25","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":25,"pages_in_order":152,"rows_per_page":100,"rows":[2401,2500],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/24","next":"/task/reinforcement-learning-1/papers/26","papers":[{"url":"/paper/m-2-dqn-a-robust-method-for-accelerating-deep","slug":"m-2-dqn-a-robust-method-for-accelerating-deep","title":"M$^2$DQN: A Robust Method for Accelerating Deep Q-learning Network","date":"2022-09-16","arxiv_id":"2209.07809","repositories_listed":1,"syntology":null},{"url":"/paper/stability-constrained-reinforcement-learning-1","slug":"stability-constrained-reinforcement-learning-1","title":"Stability Constrained Reinforcement Learning for Decentralized Real-Time Voltage Control","date":"2022-09-16","arxiv_id":"2209.07669","repositories_listed":1,"syntology":null},{"url":"/paper/toward-safe-and-accelerated-deep","slug":"toward-safe-and-accelerated-deep","title":"Toward Safe and Accelerated Deep Reinforcement Learning for Next-Generation Wireless Networks","date":"2022-09-16","arxiv_id":"2209.13532","repositories_listed":1,"syntology":null},{"url":"/paper/continuous-mdp-homomorphisms-and-homomorphic","slug":"continuous-mdp-homomorphisms-and-homomorphic","title":"Continuous MDP Homomorphisms and Homomorphic Policy Gradient","date":"2022-09-15","arxiv_id":"2209.07364","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/continuous-mdp-homomorphisms-and-homomorphic#ran","syntology_url":"https://syntology.ai/paper/2209.07364","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.07364"}},"official":{"repos":["sahandrez/homomorphic_policy_gradient"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-reuse-bias-in-off-policy-reinforcement","slug":"on-the-reuse-bias-in-off-policy-reinforcement","title":"On the Reuse Bias in Off-Policy Reinforcement Learning","date":"2022-09-15","arxiv_id":"2209.07074","repositories_listed":1,"syntology":null},{"url":"/paper/learning-state-correspondence-of","slug":"learning-state-correspondence-of","title":"Knowledge Transfer in Deep Reinforcement Learning via an RL-Specific GAN-Based Correspondence Function","date":"2022-09-14","arxiv_id":"2209.06604","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-1","slug":"deep-reinforcement-learning-for-1","title":"Deep Reinforcement Learning for Cryptocurrency Trading: Practical Approach to Address Backtest Overfitting","date":"2022-09-12","arxiv_id":"2209.05559","repositories_listed":1,"syntology":{"n":13,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/deep-reinforcement-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/2209.05559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.05559"}},"official":null}},{"url":"/paper/model-based-reinforcement-learning-with-multi","slug":"model-based-reinforcement-learning-with-multi","title":"Model-based Reinforcement Learning with Multi-step Plan Value Estimation","date":"2022-09-12","arxiv_id":"2209.05530","repositories_listed":1,"syntology":null},{"url":"/paper/unified-state-representation-learning-under","slug":"unified-state-representation-learning-under","title":"Unified State Representation Learning under Data Augmentation","date":"2022-09-12","arxiv_id":"2209.05302","repositories_listed":1,"syntology":null},{"url":"/paper/ask-before-you-act-generalising-to-novel","slug":"ask-before-you-act-generalising-to-novel","title":"Ask Before You Act: Generalising to Novel Environments by Asking Questions","date":"2022-09-10","arxiv_id":"2209.04665","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-combination-of-a-genetic-algorithm","slug":"adaptive-combination-of-a-genetic-algorithm","title":"Adaptive Combination of a Genetic Algorithm and Novelty Search for Deep Neuroevolution","date":"2022-09-08","arxiv_id":"2209.03618","repositories_listed":1,"syntology":null},{"url":"/paper/aerial-view-goal-localization-with","slug":"aerial-view-goal-localization-with","title":"Aerial View Localization with Reinforcement Learning: Towards Emulating Search-and-Rescue","date":"2022-09-08","arxiv_id":"2209.03694","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"0 ran · 3 unverified","sample_list":"/paper/aerial-view-goal-localization-with#ran","syntology_url":"https://syntology.ai/paper/2209.03694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.03694"}},"official":{"repos":["aleksispi/airloc"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/an-empirical-evaluation-of-posterior-sampling","slug":"an-empirical-evaluation-of-posterior-sampling","title":"An Empirical Evaluation of Posterior Sampling for Constrained Reinforcement Learning","date":"2022-09-08","arxiv_id":"2209.03596","repositories_listed":1,"syntology":null},{"url":"/paper/forlorn-a-framework-for-comparing-offline","slug":"forlorn-a-framework-for-comparing-offline","title":"FORLORN: A Framework for Comparing Offline Methods and Reinforcement Learning for Optimization of RAN Parameters","date":"2022-09-08","arxiv_id":"2209.13540","repositories_listed":1,"syntology":null},{"url":"/paper/q-learning-decision-transformer-leveraging","slug":"q-learning-decision-transformer-leveraging","title":"Q-learning Decision Transformer: Leveraging Dynamic Programming for Conditional Sequence Modelling in Offline RL","date":"2022-09-08","arxiv_id":"2209.03993","repositories_listed":1,"syntology":null},{"url":"/paper/reward-delay-attacks-on-deep-reinforcement","slug":"reward-delay-attacks-on-deep-reinforcement","title":"Reward Delay Attacks on Deep Reinforcement Learning","date":"2022-09-08","arxiv_id":"2209.03540","repositories_listed":1,"syntology":null},{"url":"/paper/hearts-gym-learning-reinforcement-learning-as","slug":"hearts-gym-learning-reinforcement-learning-as","title":"Hearts Gym: Learning Reinforcement Learning as a Team Event","date":"2022-09-07","arxiv_id":"2209.05466","repositories_listed":1,"syntology":null},{"url":"/paper/annealing-optimization-for-progressive","slug":"annealing-optimization-for-progressive","title":"Annealing Optimization for Progressive Learning with Stochastic Approximation","date":"2022-09-06","arxiv_id":"2209.02826","repositories_listed":1,"syntology":null},{"url":"/paper/project-proposal-a-modular-reinforcement","slug":"project-proposal-a-modular-reinforcement","title":"Project proposal: A modular reinforcement learning based automated theorem prover","date":"2022-09-06","arxiv_id":"2209.02562","repositories_listed":1,"syntology":null},{"url":"/paper/actor-prioritized-experience-replay","slug":"actor-prioritized-experience-replay","title":"Actor Prioritized Experience Replay","date":"2022-09-01","arxiv_id":"2209.00532","repositories_listed":1,"syntology":null},{"url":"/paper/intrinsic-fluctuations-of-reinforcement","slug":"intrinsic-fluctuations-of-reinforcement","title":"Intrinsic fluctuations of reinforcement learning promote cooperation","date":"2022-09-01","arxiv_id":"2209.01013","repositories_listed":1,"syntology":null},{"url":"/paper/cell-free-latent-go-explore","slug":"cell-free-latent-go-explore","title":"Cell-Free Latent Go-Explore","date":"2022-08-31","arxiv_id":"2208.14928","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-conversational-recommendations-is","slug":"rethinking-conversational-recommendations-is","title":"Rethinking Conversational Recommendations: Is Decision Tree All You Need?","date":"2022-08-31","arxiv_id":"2208.14614","repositories_listed":1,"syntology":null},{"url":"/paper/style-agnostic-reinforcement-learning","slug":"style-agnostic-reinforcement-learning","title":"Style-Agnostic Reinforcement Learning","date":"2022-08-31","arxiv_id":"2208.14863","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/style-agnostic-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2208.14863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.14863"}},"official":{"repos":["postech-cvlab/style-agnostic-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/effective-multi-user-delay-constrained","slug":"effective-multi-user-delay-constrained","title":"Effective Multi-User Delay-Constrained Scheduling with Deep Recurrent Reinforcement Learning","date":"2022-08-30","arxiv_id":"2208.14074","repositories_listed":1,"syntology":null},{"url":"/paper/goal-conditioned-q-learning-as-knowledge","slug":"goal-conditioned-q-learning-as-knowledge","title":"Goal-Conditioned Q-Learning as Knowledge Distillation","date":"2022-08-28","arxiv_id":"2208.13298","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-representation-learning-in-deep","slug":"unsupervised-representation-learning-in-deep","title":"Unsupervised Representation Learning in Deep Reinforcement Learning: A Review","date":"2022-08-27","arxiv_id":"2208.14226","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unsupervised-representation-learning-in-deep#ran","syntology_url":"https://syntology.ai/paper/2208.14226","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.14226"}},"official":{"repos":["nicob15/state_representation_learning_methods"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-comparison-of-reinforcement-learning-1","slug":"a-comparison-of-reinforcement-learning-1","title":"A Comparison of Reinforcement Learning Frameworks for Software Testing Tasks","date":"2022-08-25","arxiv_id":"2208.12136","repositories_listed":1,"syntology":null},{"url":"/paper/importance-prioritized-policy-distillation","slug":"importance-prioritized-policy-distillation","title":"Importance Prioritized Policy Distillation","date":"2022-08-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/light-weight-probing-of-unsupervised","slug":"light-weight-probing-of-unsupervised","title":"Light-weight probing of unsupervised representations for Reinforcement Learning","date":"2022-08-25","arxiv_id":"2208.12345","repositories_listed":1,"syntology":null},{"url":"/paper/variance-reduction-based-experience-replay","slug":"variance-reduction-based-experience-replay","title":"Variance Reduction based Experience Replay for Policy Optimization","date":"2022-08-25","arxiv_id":"2208.12341","repositories_listed":1,"syntology":null},{"url":"/paper/augmenting-reinforcement-learning-with-1","slug":"augmenting-reinforcement-learning-with-1","title":"Augmenting Reinforcement Learning with Transformer-based Scene Representation Learning for Decision-making of Autonomous Driving","date":"2022-08-24","arxiv_id":"2208.12263","repositories_listed":1,"syntology":null},{"url":"/paper/improving-post-processing-of-audio-event","slug":"improving-post-processing-of-audio-event","title":"Improving Post-Processing of Audio Event Detectors Using Reinforcement Learning","date":"2022-08-19","arxiv_id":"2208.09201","repositories_listed":1,"syntology":null},{"url":"/paper/a-walk-in-the-park-learning-to-walk-in-20","slug":"a-walk-in-the-park-learning-to-walk-in-20","title":"A Walk in the Park: Learning to Walk in 20 Minutes With Model-Free Reinforcement Learning","date":"2022-08-16","arxiv_id":"2208.07860","repositories_listed":1,"syntology":null},{"url":"/paper/pd-morl-preference-driven-multi-objective","slug":"pd-morl-preference-driven-multi-objective","title":"PD-MORL: Preference-Driven Multi-Objective Reinforcement Learning Algorithm","date":"2022-08-16","arxiv_id":"2208.07914","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/pd-morl-preference-driven-multi-objective#ran","syntology_url":"https://syntology.ai/paper/2208.07914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.07914"}},"official":{"repos":["tbasaklar/PDMORL-Preference-Driven-Multi-Objective-Reinforcement-Learning-Algorithm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-design-for-an-online-reinforcement","slug":"reward-design-for-an-online-reinforcement","title":"Reward Design For An Online Reinforcement Learning Algorithm Supporting Oral Self-Care","date":"2022-08-15","arxiv_id":"2208.07406","repositories_listed":1,"syntology":null},{"url":"/paper/transformer-based-value-function","slug":"transformer-based-value-function","title":"Transformer-based Value Function Decomposition for Cooperative Multi-agent Reinforcement Learning in StarCraft","date":"2022-08-15","arxiv_id":"2208.07298","repositories_listed":1,"syntology":null},{"url":"/paper/a-modular-framework-for-reinforcement","slug":"a-modular-framework-for-reinforcement","title":"A Modular Framework for Reinforcement Learning Optimal Execution","date":"2022-08-11","arxiv_id":"2208.06244","repositories_listed":1,"syntology":null},{"url":"/paper/robust-reinforcement-learning-using-offline","slug":"robust-reinforcement-learning-using-offline","title":"Robust Reinforcement Learning using Offline Data","date":"2022-08-10","arxiv_id":"2208.05129","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/robust-reinforcement-learning-using-offline#ran","syntology_url":"https://syntology.ai/paper/2208.05129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.05129"}},"official":{"repos":["zaiyan-x/RFQI"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/basis-for-intentions-efficient-inverse","slug":"basis-for-intentions-efficient-inverse","title":"Basis for Intentions: Efficient Inverse Reinforcement Learning using Past Experience","date":"2022-08-09","arxiv_id":"2208.04919","repositories_listed":1,"syntology":null},{"url":"/paper/from-scratch-to-sketch-deep-decoupled","slug":"from-scratch-to-sketch-deep-decoupled","title":"From Scratch to Sketch: Deep Decoupled Hierarchical Reinforcement Learning for Robotic Sketching Agent","date":"2022-08-09","arxiv_id":"2208.04833","repositories_listed":1,"syntology":null},{"url":"/paper/object-detection-with-deep-reinforcement","slug":"object-detection-with-deep-reinforcement","title":"Object Detection with Deep Reinforcement Learning","date":"2022-08-09","arxiv_id":"2208.04511","repositories_listed":1,"syntology":null},{"url":"/paper/socially-intelligent-genetic-agents-for-the","slug":"socially-intelligent-genetic-agents-for-the","title":"Socially Intelligent Genetic Agents for the Emergence of Explicit Norms","date":"2022-08-07","arxiv_id":"2208.03789","repositories_listed":1,"syntology":null},{"url":"/paper/a-cooperation-graph-approach-for-multiagent","slug":"a-cooperation-graph-approach-for-multiagent","title":"A Cooperation Graph Approach for Multiagent Sparse Reward Reinforcement Learning","date":"2022-08-05","arxiv_id":"2208.03002","repositories_listed":1,"syntology":null},{"url":"/paper/towards-augmented-microscopy-with","slug":"towards-augmented-microscopy-with","title":"Towards Augmented Microscopy with Reinforcement Learning-Enhanced Workflows","date":"2022-08-04","arxiv_id":"2208.02865","repositories_listed":1,"syntology":null},{"url":"/paper/mobility-aware-cooperative-caching-in","slug":"mobility-aware-cooperative-caching-in","title":"Mobility-Aware Cooperative Caching in Vehicular Edge Computing Based on Asynchronous Federated and Deep Reinforcement Learning","date":"2022-08-02","arxiv_id":"2208.01219","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-grasp-on-the-moon-from-3d-octree","slug":"learning-to-grasp-on-the-moon-from-3d-octree","title":"Learning to Grasp on the Moon from 3D Octree Observations with Deep Reinforcement Learning","date":"2022-08-01","arxiv_id":"2208.00818","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-graph-reinforcement-learning-for","slug":"model-based-graph-reinforcement-learning-for","title":"Model-based graph reinforcement learning for inductive traffic signal control","date":"2022-08-01","arxiv_id":"2208.00659","repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-correction-for-actor-critic","slug":"off-policy-correction-for-actor-critic","title":"Mitigating Off-Policy Bias in Actor-Critic Methods with One-Step Q-learning: A Novel Correction Approach","date":"2022-08-01","arxiv_id":"2208.00755","repositories_listed":1,"syntology":null},{"url":"/paper/performance-comparison-of-deep-rl-algorithms","slug":"performance-comparison-of-deep-rl-algorithms","title":"Performance Comparison of Deep RL Algorithms for Energy Systems Optimal Scheduling","date":"2022-08-01","arxiv_id":"2208.00728","repositories_listed":1,"syntology":null},{"url":"/paper/relay-hindsight-experience-replay-continual","slug":"relay-hindsight-experience-replay-continual","title":"Relay Hindsight Experience Replay: Self-Guided Continual Reinforcement Learning for Sequential Object Manipulation Tasks with Sparse Rewards","date":"2022-08-01","arxiv_id":"2208.00843","repositories_listed":1,"syntology":null},{"url":"/paper/unified-automatic-control-of-vehicular","slug":"unified-automatic-control-of-vehicular","title":"Unified Automatic Control of Vehicular Systems with Reinforcement Learning","date":"2022-07-30","arxiv_id":"2208.00268","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-ucb-provably-efficient","slug":"contrastive-ucb-provably-efficient","title":"Contrastive UCB: Provably Efficient Contrastive Self-Supervised Learning in Online Reinforcement Learning","date":"2022-07-29","arxiv_id":"2207.14800","repositories_listed":1,"syntology":null},{"url":"/paper/cyclic-policy-distillation-sample-efficient","slug":"cyclic-policy-distillation-sample-efficient","title":"Cyclic Policy Distillation: Sample-Efficient Sim-to-Real Reinforcement Learning with Domain Randomization","date":"2022-07-29","arxiv_id":"2207.14561","repositories_listed":1,"syntology":null},{"url":"/paper/sampling-attacks-on-meta-reinforcement","slug":"sampling-attacks-on-meta-reinforcement","title":"Sampling Attacks on Meta Reinforcement Learning: A Minimax Formulation and Complexity Analysis","date":"2022-07-29","arxiv_id":"2208.00081","repositories_listed":1,"syntology":null},{"url":"/paper/safe-and-robust-experience-sharing-for","slug":"safe-and-robust-experience-sharing-for","title":"Safe and Robust Experience Sharing for Deterministic Policy Gradient Algorithms","date":"2022-07-27","arxiv_id":"2207.13453","repositories_listed":1,"syntology":null},{"url":"/paper/learning-bipedal-walking-on-planned-footsteps","slug":"learning-bipedal-walking-on-planned-footsteps","title":"Learning Bipedal Walking On Planned Footsteps For Humanoid Robots","date":"2022-07-26","arxiv_id":"2207.12644","repositories_listed":1,"syntology":null},{"url":"/paper/lifelong-machine-learning-of-functionally","slug":"lifelong-machine-learning-of-functionally","title":"Lifelong Machine Learning of Functionally Compositional Structures","date":"2022-07-25","arxiv_id":"2207.12256","repositories_listed":1,"syntology":null},{"url":"/paper/live-in-the-moment-learning-dynamics-model","slug":"live-in-the-moment-learning-dynamics-model","title":"Live in the Moment: Learning Dynamics Model Adapted to Evolving Policy","date":"2022-07-25","arxiv_id":"2207.12141","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/live-in-the-moment-learning-dynamics-model#ran","syntology_url":"https://syntology.ai/paper/2207.12141","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.12141"}},"official":{"repos":["si0wang/pdml"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/post-processing-networks-method-for","slug":"post-processing-networks-method-for","title":"Post-processing Networks: Method for Optimizing Pipeline Task-oriented Dialogue Systems using Reinforcement Learning","date":"2022-07-25","arxiv_id":"2207.12185","repositories_listed":1,"syntology":null},{"url":"/paper/learning-soccer-juggling-skills-with-layer","slug":"learning-soccer-juggling-skills-with-layer","title":"Learning Soccer Juggling Skills with Layer-wise Mixture-of-Experts","date":"2022-07-24","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/driver-dojo-a-benchmark-for-generalizable","slug":"driver-dojo-a-benchmark-for-generalizable","title":"Driver Dojo: A Benchmark for Generalizable Reinforcement Learning for Autonomous Driving","date":"2022-07-23","arxiv_id":"2207.11432","repositories_listed":1,"syntology":null},{"url":"/paper/robust-knowledge-adaptation-for-dynamic-graph","slug":"robust-knowledge-adaptation-for-dynamic-graph","title":"Robust Knowledge Adaptation for Dynamic Graph Neural Networks","date":"2022-07-22","arxiv_id":"2207.10839","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-implementation-of-a-reinforcement","slug":"on-the-implementation-of-a-reinforcement","title":"On the Implementation of a Reinforcement Learning-based Capacity Sharing Algorithm in O-RAN","date":"2022-07-21","arxiv_id":"2207.10390","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-energies-of-the","slug":"reinforcement-learning-for-energies-of-the","title":"Reinforcement learning for Energies of the future and carbon neutrality: a Challenge Design","date":"2022-07-21","arxiv_id":"2207.10330","repositories_listed":1,"syntology":null},{"url":"/paper/solving-the-optimal-stopping-problem-with","slug":"solving-the-optimal-stopping-problem-with","title":"Solving the optimal stopping problem with reinforcement learning: an application in financial option exercise","date":"2022-07-21","arxiv_id":"2208.00765","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-market-making-1","slug":"deep-reinforcement-learning-for-market-making-1","title":"Deep Reinforcement Learning for Market Making Under a Hawkes Process-Based Limit Order Book Model","date":"2022-07-20","arxiv_id":"2207.09951","repositories_listed":1,"syntology":null},{"url":"/paper/successor-representation-active-inference","slug":"successor-representation-active-inference","title":"Successor Representation Active Inference","date":"2022-07-20","arxiv_id":"2207.09897","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/successor-representation-active-inference#ran","syntology_url":"https://syntology.ai/paper/2207.09897","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.09897"}},"official":{"repos":["BerenMillidge/Active_Inference_Successor_Representations"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/generalizing-goal-conditioned-reinforcement","slug":"generalizing-goal-conditioned-reinforcement","title":"Generalizing Goal-Conditioned Reinforcement Learning with Variational Causal Reasoning","date":"2022-07-19","arxiv_id":"2207.09081","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalizing-goal-conditioned-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2207.09081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.09081"}},"official":{"repos":["gilgameshd/grader"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/magpie-automatically-tuning-static-parameters","slug":"magpie-automatically-tuning-static-parameters","title":"Magpie: Automatically Tuning Static Parameters for Distributed File Systems using Deep Reinforcement Learning","date":"2022-07-19","arxiv_id":"2207.09298","repositories_listed":1,"syntology":null},{"url":"/paper/a-meta-reinforcement-learning-algorithm-for","slug":"a-meta-reinforcement-learning-algorithm-for","title":"A Meta-Reinforcement Learning Algorithm for Causal Discovery","date":"2022-07-18","arxiv_id":"2207.08457","repositories_listed":1,"syntology":null},{"url":"/paper/active-exploration-for-inverse-reinforcement","slug":"active-exploration-for-inverse-reinforcement","title":"Active Exploration for Inverse Reinforcement Learning","date":"2022-07-18","arxiv_id":"2207.08645","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":3,"n_ran_checked":3,"n_instrument":7,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"10 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 7 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/active-exploration-for-inverse-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2207.08645","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.08645"}},"official":{"repos":["lasgroup/aceirl"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bootstrap-state-representation-using-style","slug":"bootstrap-state-representation-using-style","title":"Bootstrap State Representation using Style Transfer for Better Generalization in Deep Reinforcement Learning","date":"2022-07-15","arxiv_id":"2207.07749","repositories_listed":1,"syntology":null},{"url":"/paper/asset-allocation-from-markowitz-to-deep","slug":"asset-allocation-from-markowitz-to-deep","title":"Asset Allocation: From Markowitz to Deep Reinforcement Learning","date":"2022-07-14","arxiv_id":"2208.07158","repositories_listed":1,"syntology":null},{"url":"/paper/a-general-contextualized-rewriting-framework","slug":"a-general-contextualized-rewriting-framework","title":"A General Contextualized Rewriting Framework for Text Summarization","date":"2022-07-13","arxiv_id":"2207.05948","repositories_listed":1,"syntology":null},{"url":"/paper/brick-tic-tac-toe-exploring-the","slug":"brick-tic-tac-toe-exploring-the","title":"Brick Tic-Tac-Toe: Exploring the Generalizability of AlphaZero to Novel Test Environments","date":"2022-07-13","arxiv_id":"2207.05991","repositories_listed":1,"syntology":null},{"url":"/paper/hindsight-learning-for-mdps-with-exogenous","slug":"hindsight-learning-for-mdps-with-exogenous","title":"Hindsight Learning for MDPs with Exogenous Inputs","date":"2022-07-13","arxiv_id":"2207.06272","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-assisted-recursive","slug":"reinforcement-learning-assisted-recursive","title":"Reinforcement Learning Assisted Recursive QAOA","date":"2022-07-13","arxiv_id":"2207.06294","repositories_listed":1,"syntology":null},{"url":"/paper/learning-bellman-complete-representations-for","slug":"learning-bellman-complete-representations-for","title":"Learning Bellman Complete Representations for Offline Policy Evaluation","date":"2022-07-12","arxiv_id":"2207.05837","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/learning-bellman-complete-representations-for#ran","syntology_url":"https://syntology.ai/paper/2207.05837","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05837"}},"official":{"repos":["causalml/bcrl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/online-game-level-generation-from-music","slug":"online-game-level-generation-from-music","title":"Online Game Level Generation from Music","date":"2022-07-12","arxiv_id":"2207.05271","repositories_listed":1,"syntology":null},{"url":"/paper/reactive-exploration-to-cope-with-non","slug":"reactive-exploration-to-cope-with-non","title":"Reactive Exploration to Cope with Non-Stationarity in Lifelong Reinforcement Learning","date":"2022-07-12","arxiv_id":"2207.05742","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-disentanglement-of-representations","slug":"temporal-disentanglement-of-representations","title":"Temporal Disentanglement of Representations for Improved Generalisation in Reinforcement Learning","date":"2022-07-12","arxiv_id":"2207.05480","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/temporal-disentanglement-of-representations#ran","syntology_url":"https://syntology.ai/paper/2207.05480","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05480"}},"official":{"repos":["uoe-agents/ted"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/grounding-aleatoric-uncertainty-in-1","slug":"grounding-aleatoric-uncertainty-in-1","title":"Grounding Aleatoric Uncertainty for Unsupervised Environment Design","date":"2022-07-11","arxiv_id":"2207.05219","repositories_listed":1,"syntology":null},{"url":"/paper/composuite-a-compositional-reinforcement","slug":"composuite-a-compositional-reinforcement","title":"CompoSuite: A Compositional Reinforcement Learning Benchmark","date":"2022-07-08","arxiv_id":"2207.04136","repositories_listed":1,"syntology":null},{"url":"/paper/interaction-pattern-disentangling-for-multi","slug":"interaction-pattern-disentangling-for-multi","title":"Interaction Pattern Disentangling for Multi-Agent Reinforcement Learning","date":"2022-07-08","arxiv_id":"2207.03902","repositories_listed":1,"syntology":null},{"url":"/paper/reinforced-lin-kernighan-helsgaun-algorithms","slug":"reinforced-lin-kernighan-helsgaun-algorithms","title":"Reinforced Lin-Kernighan-Helsgaun Algorithms for the Traveling Salesman Problems","date":"2022-07-08","arxiv_id":"2207.03876","repositories_listed":1,"syntology":null},{"url":"/paper/storehouse-a-reinforcement-learning","slug":"storehouse-a-reinforcement-learning","title":"Storehouse: a Reinforcement Learning Environment for Optimizing Warehouse Management","date":"2022-07-08","arxiv_id":"2207.03851","repositories_listed":1,"syntology":null},{"url":"/paper/robust-optimal-well-control-using-an-adaptive","slug":"robust-optimal-well-control-using-an-adaptive","title":"Robust optimal well control using an adaptive multi-grid reinforcement learning framework","date":"2022-07-07","arxiv_id":"2207.03253","repositories_listed":1,"syntology":null},{"url":"/paper/stochastic-optimal-well-control-in-subsurface","slug":"stochastic-optimal-well-control-in-subsurface","title":"Stochastic optimal well control in subsurface reservoirs using reinforcement learning","date":"2022-07-07","arxiv_id":"2207.03456","repositories_listed":1,"syntology":null},{"url":"/paper/a-learning-system-for-motion-planning-of-free","slug":"a-learning-system-for-motion-planning-of-free","title":"A Learning System for Motion Planning of Free-Float Dual-Arm Space Manipulator towards Non-Cooperative Object","date":"2022-07-06","arxiv_id":"2207.02464","repositories_listed":1,"syntology":null},{"url":"/paper/implementing-reinforcement-learning-1","slug":"implementing-reinforcement-learning-1","title":"Implementing Reinforcement Learning Datacenter Congestion Control in NVIDIA NICs","date":"2022-07-05","arxiv_id":"2207.02295","repositories_listed":1,"syntology":null},{"url":"/paper/learning-task-embeddings-for-teamwork","slug":"learning-task-embeddings-for-teamwork","title":"Learning Task Embeddings for Teamwork Adaptation in Multi-Agent Reinforcement Learning","date":"2022-07-05","arxiv_id":"2207.02249","repositories_listed":1,"syntology":null},{"url":"/paper/robust-reinforcement-learning-in-continuous","slug":"robust-reinforcement-learning-in-continuous","title":"Robust Reinforcement Learning in Continuous Control Tasks with Uncertainty Set Regularization","date":"2022-07-05","arxiv_id":"2207.02016","repositories_listed":1,"syntology":null},{"url":"/paper/general-policy-evaluation-and-improvement-by","slug":"general-policy-evaluation-and-improvement-by","title":"General Policy Evaluation and Improvement by Learning to Identify Few But Crucial States","date":"2022-07-04","arxiv_id":"2207.01566","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/general-policy-evaluation-and-improvement-by#ran","syntology_url":"https://syntology.ai/paper/2207.01566","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.01566"}},"official":{"repos":["idsia/policyevaluator"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/solving-the-traveling-salesperson-problem","slug":"solving-the-traveling-salesperson-problem","title":"Solving the Traveling Salesperson Problem with Precedence Constraints by Deep Reinforcement Learning","date":"2022-07-04","arxiv_id":"2207.01443","repositories_listed":1,"syntology":null},{"url":"/paper/renaissance-robot-optimal-transport-policy","slug":"renaissance-robot-optimal-transport-policy","title":"Renaissance Robot: Optimal Transport Policy Fusion for Learning Diverse Skills","date":"2022-07-03","arxiv_id":"2207.00978","repositories_listed":1,"syntology":null},{"url":"/paper/stabilizing-off-policy-deep-reinforcement","slug":"stabilizing-off-policy-deep-reinforcement","title":"Stabilizing Off-Policy Deep Reinforcement Learning from Pixels","date":"2022-07-03","arxiv_id":"2207.00986","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/stabilizing-off-policy-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2207.00986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.00986"}},"official":{"repos":["aladoro/stabilizing-off-policy-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/usher-unbiased-sampling-for-hindsight","slug":"usher-unbiased-sampling-for-hindsight","title":"USHER: Unbiased Sampling for Hindsight Experience Replay","date":"2022-07-03","arxiv_id":"2207.01115","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/usher-unbiased-sampling-for-hindsight#ran","syntology_url":"https://syntology.ai/paper/2207.01115","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.01115"}},"official":null}},{"url":"/paper/interactive-query-assisted-summarization-via","slug":"interactive-query-assisted-summarization-via","title":"Interactive Query-Assisted Summarization via Deep Reinforcement Learning","date":"2022-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-natural-language-generation-with","slug":"learning-natural-language-generation-with","title":"Learning Natural Language Generation with Truncated Reinforcement Learning","date":"2022-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null}],"record_sha256":"325a1f04274b66a75ea855ddb6feceab31cd877195962f2747228012c434127f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}