{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/33","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":33,"pages_in_order":152,"rows_per_page":100,"rows":[3201,3300],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/32","next":"/task/reinforcement-learning-1/papers/34","papers":[{"url":"/paper/simplifying-deep-reinforcement-learning-via","slug":"simplifying-deep-reinforcement-learning-via","title":"Simplifying Deep Reinforcement Learning via Self-Supervision","date":"2021-06-10","arxiv_id":"2106.05526","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/simplifying-deep-reinforcement-learning-via#ran","syntology_url":"https://syntology.ai/paper/2106.05526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.05526"}},"official":{"repos":["daochenzha/SSRL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/synthesising-reinforcement-learning-policies","slug":"synthesising-reinforcement-learning-policies","title":"Synthesising Reinforcement Learning Policies through Set-Valued Inductive Rule Learning","date":"2021-06-10","arxiv_id":"2106.06009","repositories_listed":1,"syntology":null},{"url":"/paper/pretrained-encoders-are-all-you-need","slug":"pretrained-encoders-are-all-you-need","title":"Pretrained Encoders are All You Need","date":"2021-06-09","arxiv_id":"2106.05139","repositories_listed":1,"syntology":null},{"url":"/paper/pretraining-representations-for-data","slug":"pretraining-representations-for-data","title":"Pretraining Representations for Data-Efficient Reinforcement Learning","date":"2021-06-09","arxiv_id":"2106.04799","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pretraining-representations-for-data#ran","syntology_url":"https://syntology.ai/paper/2106.04799","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04799"}},"official":{"repos":["mila-iqia/SGI"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/self-paced-context-evaluation-for-contextual","slug":"self-paced-context-evaluation-for-contextual","title":"Self-Paced Context Evaluation for Contextual Reinforcement Learning","date":"2021-06-09","arxiv_id":"2106.05110","repositories_listed":1,"syntology":null},{"url":"/paper/who-is-the-strongest-enemy-towards-optimal","slug":"who-is-the-strongest-enemy-towards-optimal","title":"Who Is the Strongest Enemy? Towards Optimal and Efficient Evasion Attacks in Deep RL","date":"2021-06-09","arxiv_id":"2106.05087","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":1,"n_instrument":8,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 8 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/who-is-the-strongest-enemy-towards-optimal#ran","syntology_url":"https://syntology.ai/paper/2106.05087","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.05087"}},"official":{"repos":["umd-huang-lab/paad_adv_rl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/curriculum-design-for-teaching-via","slug":"curriculum-design-for-teaching-via","title":"Curriculum Design for Teaching via Demonstrations: Theory and Applications","date":"2021-06-08","arxiv_id":"2106.04696","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":1,"n_ran_checked":2,"n_instrument":6,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":9,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/curriculum-design-for-teaching-via#ran","syntology_url":"https://syntology.ai/paper/2106.04696","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04696"}},"official":{"repos":["adishs/neurips2021_curriculum-teaching-demonstrations_code"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-sparse-training-for-deep","slug":"dynamic-sparse-training-for-deep","title":"Dynamic Sparse Training for Deep Reinforcement Learning","date":"2021-06-08","arxiv_id":"2106.04217","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-sparse-training-for-deep#ran","syntology_url":"https://syntology.ai/paper/2106.04217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04217"}},"official":{"repos":["GhadaSokar/Dynamic-Sparse-Training-for-Deep-Reinforcement-Learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-markov-state-abstractions-for-deep","slug":"learning-markov-state-abstractions-for-deep","title":"Learning Markov State Abstractions for Deep Reinforcement Learning","date":"2021-06-08","arxiv_id":"2106.04379","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-markov-state-abstractions-for-deep#ran","syntology_url":"https://syntology.ai/paper/2106.04379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04379"}},"official":{"repos":["camall3n/markov-state-abstractions"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/believe-what-you-see-implicit-constraint","slug":"believe-what-you-see-implicit-constraint","title":"Believe What You See: Implicit Constraint Approach for Offline Multi-Agent Reinforcement Learning","date":"2021-06-07","arxiv_id":"2106.03400","repositories_listed":1,"syntology":null},{"url":"/paper/correcting-momentum-in-temporal-difference-1","slug":"correcting-momentum-in-temporal-difference-1","title":"Correcting Momentum in Temporal Difference Learning","date":"2021-06-07","arxiv_id":"2106.03955","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/correcting-momentum-in-temporal-difference-1#ran","syntology_url":"https://syntology.ai/paper/2106.03955","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03955"}},"official":{"repos":["bengioe/staleness-corrected-momentum"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/task-driven-semantic-coding-via-reinforcement","slug":"task-driven-semantic-coding-via-reinforcement","title":"Task-driven Semantic Coding via Reinforcement Learning","date":"2021-06-07","arxiv_id":"2106.03511","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/task-driven-semantic-coding-via-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2106.03511","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03511"}},"official":{"repos":["USTC-IMCL/Task-driven-Semantic-Coding-via-RL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/verifiable-and-compositional-reinforcement","slug":"verifiable-and-compositional-reinforcement","title":"Verifiable and Compositional Reinforcement Learning Systems","date":"2021-06-07","arxiv_id":"2106.05864","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/verifiable-and-compositional-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2106.05864","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.05864"}},"official":{"repos":["cyrusneary/verifiable-compositional-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/xirl-cross-embodiment-inverse-reinforcement","slug":"xirl-cross-embodiment-inverse-reinforcement","title":"XIRL: Cross-embodiment Inverse Reinforcement Learning","date":"2021-06-07","arxiv_id":"2106.03911","repositories_listed":1,"syntology":null},{"url":"/paper/control-oriented-model-based-reinforcement","slug":"control-oriented-model-based-reinforcement","title":"Control-Oriented Model-Based Reinforcement Learning with Implicit Differentiation","date":"2021-06-06","arxiv_id":"2106.03273","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/control-oriented-model-based-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2106.03273","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03273"}},"official":{"repos":["evgenii-nikishin/omd"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/distributional-reinforcement-learning-with-4","slug":"distributional-reinforcement-learning-with-4","title":"Distributional Reinforcement Learning with Unconstrained Monotonic Neural Networks","date":"2021-06-06","arxiv_id":"2106.03228","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-continuous-control-with-double","slug":"efficient-continuous-control-with-double","title":"Efficient Continuous Control with Double Actors and Regularized Critics","date":"2021-06-06","arxiv_id":"2106.03050","repositories_listed":1,"syntology":null},{"url":"/paper/schedulenet-learn-to-solve-multi-agent","slug":"schedulenet-learn-to-solve-multi-agent","title":"ScheduleNet: Learn to solve multi-agent scheduling problems with reinforcement learning","date":"2021-06-06","arxiv_id":"2106.03051","repositories_listed":1,"syntology":null},{"url":"/paper/malib-a-parallel-framework-for-population","slug":"malib-a-parallel-framework-for-population","title":"MALib: A Parallel Framework for Population-based Multi-agent Reinforcement Learning","date":"2021-06-05","arxiv_id":"2106.07551","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/malib-a-parallel-framework-for-population#ran","syntology_url":"https://syntology.ai/paper/2106.07551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.07551"}},"official":{"repos":["sjtu-marl/malib"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/same-state-different-task-continual","slug":"same-state-different-task-continual","title":"Same State, Different Task: Continual Reinforcement Learning without Interference","date":"2021-06-05","arxiv_id":"2106.02940","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/same-state-different-task-continual#ran","syntology_url":"https://syntology.ai/paper/2106.02940","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.02940"}},"official":{"repos":["skezle/owl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-trajectory-representation-learning-for","slug":"cross-trajectory-representation-learning-for","title":"Cross-Trajectory Representation Learning for Zero-Shot Generalization in RL","date":"2021-06-04","arxiv_id":"2106.02193","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":2,"n_ran_checked":6,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"7 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/cross-trajectory-representation-learning-for#ran","syntology_url":"https://syntology.ai/paper/2106.02193","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.02193"}},"official":{"repos":["bmazoure/ctrl_public"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":2,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/model-agnostic-and-scalable-counterfactual","slug":"model-agnostic-and-scalable-counterfactual","title":"Model-agnostic and Scalable Counterfactual Explanations via Reinforcement Learning","date":"2021-06-04","arxiv_id":"2106.02597","repositories_listed":1,"syntology":null},{"url":"/paper/online-reinforcement-learning-with-sparse","slug":"online-reinforcement-learning-with-sparse","title":"Online reinforcement learning with sparse rewards through an active inference capsule","date":"2021-06-04","arxiv_id":"2106.02390","repositories_listed":1,"syntology":null},{"url":"/paper/rl-darts-differentiable-architecture-search","slug":"rl-darts-differentiable-architecture-search","title":"Differentiable Architecture Search for Reinforcement Learning","date":"2021-06-04","arxiv_id":"2106.02229","repositories_listed":1,"syntology":null},{"url":"/paper/a-consciousness-inspired-planning-agent-for","slug":"a-consciousness-inspired-planning-agent-for","title":"A Consciousness-Inspired Planning Agent for Model-Based Reinforcement Learning","date":"2021-06-03","arxiv_id":"2106.02097","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":3,"n_ran_checked":3,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/a-consciousness-inspired-planning-agent-for#ran","syntology_url":"https://syntology.ai/paper/2106.02097","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.02097"}},"official":{"repos":["mila-iqia/conscious-planning"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/optimization-based-algebraic-multigrid","slug":"optimization-based-algebraic-multigrid","title":"Optimization-Based Algebraic Multigrid Coarsening Using Reinforcement Learning","date":"2021-06-03","arxiv_id":"2106.01854","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimization-based-algebraic-multigrid#ran","syntology_url":"https://syntology.ai/paper/2106.01854","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.01854"}},"official":{"repos":["compdyn/rl_grid_coarsen"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-learning-to-play-piano-with-dexterous","slug":"towards-learning-to-play-piano-with-dexterous","title":"Towards Learning to Play Piano with Dexterous Hands and Touch","date":"2021-06-03","arxiv_id":"2106.02040","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-in-quantitative","slug":"deep-reinforcement-learning-in-quantitative","title":"Deep Reinforcement Learning in Quantitative Algorithmic Trading: A Review","date":"2021-05-31","arxiv_id":"2106.00123","repositories_listed":1,"syntology":null},{"url":"/paper/q-attention-enabling-efficient-learning-for","slug":"q-attention-enabling-efficient-learning-for","title":"Q-attention: Enabling Efficient Learning for Vision-based Robotic Manipulation","date":"2021-05-31","arxiv_id":"2105.14829","repositories_listed":1,"syntology":null},{"url":"/paper/a-nearly-blackwell-optimal-policy-gradient","slug":"a-nearly-blackwell-optimal-policy-gradient","title":"A nearly Blackwell-optimal policy gradient method","date":"2021-05-28","arxiv_id":"2105.13609","repositories_listed":1,"syntology":null},{"url":"/paper/improving-generalization-in-meta-rl-with","slug":"improving-generalization-in-meta-rl-with","title":"Improving Generalization in Meta-RL with Imaginary Tasks from Latent Dynamics Mixture","date":"2021-05-28","arxiv_id":"2105.13524","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/improving-generalization-in-meta-rl-with#ran","syntology_url":"https://syntology.ai/paper/2105.13524","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.13524"}},"official":{"repos":["suyoung-lee/ldm"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/adversarial-intrinsic-motivation-for","slug":"adversarial-intrinsic-motivation-for","title":"Adversarial Intrinsic Motivation for Reinforcement Learning","date":"2021-05-27","arxiv_id":"2105.13345","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/adversarial-intrinsic-motivation-for#ran","syntology_url":"https://syntology.ai/paper/2105.13345","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.13345"}},"official":{"repos":["iDurugkar/adversarial-intrinsic-motivation"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/successive-convex-approximation-based-off","slug":"successive-convex-approximation-based-off","title":"Successive Convex Approximation Based Off-Policy Optimization for Constrained Reinforcement Learning","date":"2021-05-26","arxiv_id":"2105.12545","repositories_listed":1,"syntology":null},{"url":"/paper/a-comparison-of-reward-functions-in-q","slug":"a-comparison-of-reward-functions-in-q","title":"A Comparison of Reward Functions in Q-Learning Applied to a Cart Position Problem","date":"2021-05-25","arxiv_id":"2105.11617","repositories_listed":1,"syntology":null},{"url":"/paper/robust-value-iteration-for-continuous-control","slug":"robust-value-iteration-for-continuous-control","title":"Robust Value Iteration for Continuous Control Tasks","date":"2021-05-25","arxiv_id":"2105.12189","repositories_listed":1,"syntology":null},{"url":"/paper/towards-scalable-verification-of-rl-driven","slug":"towards-scalable-verification-of-rl-driven","title":"Towards Scalable Verification of Deep Reinforcement Learning","date":"2021-05-25","arxiv_id":"2105.11931","repositories_listed":1,"syntology":null},{"url":"/paper/an-efficient-application-of-neuroevolution","slug":"an-efficient-application-of-neuroevolution","title":"An Efficient Application of Neuroevolution for Competitive Multiagent Learning","date":"2021-05-23","arxiv_id":"2105.10907","repositories_listed":1,"syntology":null},{"url":"/paper/continual-world-a-robotic-benchmark-for","slug":"continual-world-a-robotic-benchmark-for","title":"Continual World: A Robotic Benchmark For Continual Reinforcement Learning","date":"2021-05-23","arxiv_id":"2105.10919","repositories_listed":1,"syntology":null},{"url":"/paper/certification-of-iterative-predictions-in","slug":"certification-of-iterative-predictions-in","title":"Certification of Iterative Predictions in Bayesian Neural Networks","date":"2021-05-21","arxiv_id":"2105.10134","repositories_listed":1,"syntology":null},{"url":"/paper/cooperative-multi-agent-reinforcement-5","slug":"cooperative-multi-agent-reinforcement-5","title":"Cooperative Multi-Agent Reinforcement Learning with Sequential Credit Assignment","date":"2021-05-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/ensemble-quantile-networks-uncertainty-aware","slug":"ensemble-quantile-networks-uncertainty-aware","title":"Ensemble Quantile Networks: Uncertainty-Aware Reinforcement Learning with Applications in Autonomous Driving","date":"2021-05-21","arxiv_id":"2105.10266","repositories_listed":1,"syntology":null},{"url":"/paper/offline-meta-reinforcement-learning-1","slug":"offline-meta-reinforcement-learning-1","title":"Offline Meta Reinforcement Learning -- Identifiability Challenges and Effective Data Collection Strategies","date":"2021-05-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/on-instrumental-variable-regression-for-deep","slug":"on-instrumental-variable-regression-for-deep","title":"On Instrumental Variable Regression for Deep Offline Policy Evaluation","date":"2021-05-21","arxiv_id":"2105.10148","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-instrumental-variable-regression-for-deep#ran","syntology_url":"https://syntology.ai/paper/2105.10148","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.10148"}},"official":{"repos":["liyuan9988/IVOPEwithACME"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rule-augmented-unsupervised-constituency","slug":"rule-augmented-unsupervised-constituency","title":"Rule Augmented Unsupervised Constituency Parsing","date":"2021-05-21","arxiv_id":"2105.10193","repositories_listed":1,"syntology":null},{"url":"/paper/minimum-delay-adaptation-in-non-stationary","slug":"minimum-delay-adaptation-in-non-stationary","title":"Minimum-Delay Adaptation in Non-Stationary Reinforcement Learning via Online High-Confidence Change-Point Detection","date":"2021-05-20","arxiv_id":"2105.09452","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/minimum-delay-adaptation-in-non-stationary#ran","syntology_url":"https://syntology.ai/paper/2105.09452","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.09452"}},"official":{"repos":["LucasAlegre/mbcd"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-optimal-3","slug":"deep-reinforcement-learning-for-optimal-3","title":"Deep Reinforcement Learning for Optimal Stopping with Application in Financial Engineering","date":"2021-05-19","arxiv_id":"2105.08877","repositories_listed":1,"syntology":null},{"url":"/paper/enforcing-policy-feasibility-constraints","slug":"enforcing-policy-feasibility-constraints","title":"Enforcing Policy Feasibility Constraints through Differentiable Projection for Energy Optimization","date":"2021-05-19","arxiv_id":"2105.08881","repositories_listed":1,"syntology":null},{"url":"/paper/improved-exploring-starts-by-kernel-density","slug":"improved-exploring-starts-by-kernel-density","title":"Improved Exploring Starts by Kernel Density Estimation-Based State-Space Coverage Acceleration in Reinforcement Learning","date":"2021-05-19","arxiv_id":"2105.08990","repositories_listed":1,"syntology":null},{"url":"/paper/coach-player-multi-agent-reinforcement","slug":"coach-player-multi-agent-reinforcement","title":"Coach-Player Multi-Agent Reinforcement Learning for Dynamic Team Composition","date":"2021-05-18","arxiv_id":"2105.08692","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coach-player-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2105.08692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.08692"}},"official":{"repos":["cranial-xix/marl-copa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-reinforcement-learning-by-tracking-task","slug":"meta-reinforcement-learning-by-tracking-task","title":"Meta-Reinforcement Learning by Tracking Task Non-stationarity","date":"2021-05-18","arxiv_id":"2105.08834","repositories_listed":1,"syntology":null},{"url":"/paper/behavior-based-neuroevolutionary-training-in","slug":"behavior-based-neuroevolutionary-training-in","title":"Behavior-based Neuroevolutionary Training in Reinforcement Learning","date":"2021-05-17","arxiv_id":"2105.07960","repositories_listed":1,"syntology":null},{"url":"/paper/generic-itemset-mining-based-on-reinforcement","slug":"generic-itemset-mining-based-on-reinforcement","title":"Generic Itemset Mining Based on Reinforcement Learning","date":"2021-05-17","arxiv_id":"2105.07753","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-offline-planning-with-trajectory","slug":"model-based-offline-planning-with-trajectory","title":"Model-Based Offline Planning with Trajectory Pruning","date":"2021-05-16","arxiv_id":"2105.07351","repositories_listed":1,"syntology":null},{"url":"/paper/regret-minimization-experience-replay","slug":"regret-minimization-experience-replay","title":"Regret Minimization Experience Replay in Off-Policy Reinforcement Learning","date":"2021-05-15","arxiv_id":"2105.07253","repositories_listed":1,"syntology":null},{"url":"/paper/ordering-based-causal-discovery-with-1","slug":"ordering-based-causal-discovery-with-1","title":"Ordering-Based Causal Discovery with Reinforcement Learning","date":"2021-05-14","arxiv_id":"2105.06631","repositories_listed":1,"syntology":null},{"url":"/paper/principled-exploration-via-optimistic","slug":"principled-exploration-via-optimistic","title":"Principled Exploration via Optimistic Bootstrapping and Backward Induction","date":"2021-05-13","arxiv_id":"2105.06022","repositories_listed":1,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":12,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/principled-exploration-via-optimistic#ran","syntology_url":"https://syntology.ai/paper/2105.06022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.06022"}},"official":{"repos":["Baichenjia/OB2I"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/a-reinforcement-learning-environment-for-1","slug":"a-reinforcement-learning-environment-for-1","title":"A Reinforcement Learning Environment for Multi-Service UAV-enabled Wireless Systems","date":"2021-05-11","arxiv_id":"2105.05094","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-from-reformulations-in","slug":"reinforcement-learning-from-reformulations-in","title":"Reinforcement Learning from Reformulations in Conversational Question Answering over Knowledge Graphs","date":"2021-05-11","arxiv_id":"2105.04850","repositories_listed":1,"syntology":null},{"url":"/paper/spectral-normalisation-for-deep-reinforcement","slug":"spectral-normalisation-for-deep-reinforcement","title":"Spectral Normalisation for Deep Reinforcement Learning: an Optimisation Perspective","date":"2021-05-11","arxiv_id":"2105.05246","repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-reinforcement-learning-approach-to-3","slug":"a-deep-reinforcement-learning-approach-to-3","title":"A Deep Reinforcement Learning Approach to Audio-Based Navigation in a Multi-Speaker Environment","date":"2021-05-10","arxiv_id":"2105.04488","repositories_listed":1,"syntology":null},{"url":"/paper/lightweight-image-super-resolution-with-3","slug":"lightweight-image-super-resolution-with-3","title":"Differentiable Neural Architecture Search for Extremely Lightweight Image Super-Resolution","date":"2021-05-09","arxiv_id":"2105.03939","repositories_listed":1,"syntology":null},{"url":"/paper/deeprf-deep-reinforcement-learning-designed","slug":"deeprf-deep-reinforcement-learning-designed","title":"Deep reinforcement learning-designed radiofrequency waveform in MRI","date":"2021-05-07","arxiv_id":"2105.03061","repositories_listed":1,"syntology":null},{"url":"/paper/evening-the-score-targeting-sars-cov-2","slug":"evening-the-score-targeting-sars-cov-2","title":"Evening the Score: Targeting SARS-CoV-2 Protease Inhibition in Graph Generative Models for Therapeutic Candidates","date":"2021-05-07","arxiv_id":"2105.10489","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-multi-agent-policy-optimization","slug":"model-based-multi-agent-policy-optimization","title":"Model-based Multi-agent Policy Optimization with Adaptive Opponent-wise Rollouts","date":"2021-05-07","arxiv_id":"2105.03363","repositories_listed":1,"syntology":null},{"url":"/paper/meta-learning-based-deep-reinforcement","slug":"meta-learning-based-deep-reinforcement","title":"Meta-Learning-Based Deep Reinforcement Learning for Multiobjective Optimization Problems","date":"2021-05-06","arxiv_id":"2105.02741","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/meta-learning-based-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2105.02741","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.02741"}},"official":{"repos":["zhangzizhen/ml-dam"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/model-free-policy-evaluation-in-reinforcement","slug":"model-free-policy-evaluation-in-reinforcement","title":"UVIP: Model-Free Approach to Evaluate Reinforcement Learning Algorithms","date":"2021-05-05","arxiv_id":"2105.02135","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-adaptive-3","slug":"deep-reinforcement-learning-for-adaptive-3","title":"Deep Reinforcement Learning for Adaptive Exploration of Unknown Environments","date":"2021-05-04","arxiv_id":"2105.01606","repositories_listed":1,"syntology":null},{"url":"/paper/robotic-surgery-with-lean-reinforcement","slug":"robotic-surgery-with-lean-reinforcement","title":"Robotic Surgery With Lean Reinforcement Learning","date":"2021-05-03","arxiv_id":"2105.01006","repositories_listed":1,"syntology":null},{"url":"/paper/curious-exploration-and-return-based-memory","slug":"curious-exploration-and-return-based-memory","title":"Curious Exploration and Return-based Memory Restoration for Deep Reinforcement Learning","date":"2021-05-02","arxiv_id":"2105.00499","repositories_listed":1,"syntology":null},{"url":"/paper/adapting-to-reward-progressivity-via-spectral-1","slug":"adapting-to-reward-progressivity-via-spectral-1","title":"Adapting to Reward Progressivity via Spectral Reinforcement Learning","date":"2021-04-29","arxiv_id":"2104.14138","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/adapting-to-reward-progressivity-via-spectral-1#ran","syntology_url":"https://syntology.ai/paper/2104.14138","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.14138"}},"official":{"repos":["mchldann/SpectralDQN"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-scalable-and-reproducible-system-on-chip","slug":"a-scalable-and-reproducible-system-on-chip","title":"A Scalable and Reproducible System-on-Chip Simulation for Reinforcement Learning","date":"2021-04-27","arxiv_id":"2104.13187","repositories_listed":1,"syntology":null},{"url":"/paper/computational-performance-of-deep","slug":"computational-performance-of-deep","title":"Computational Performance of Deep Reinforcement Learning to find Nash Equilibria","date":"2021-04-26","arxiv_id":"2104.12895","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-grasping-policies-for-human-in-the","slug":"end-to-end-grasping-policies-for-human-in-the","title":"End-to-end grasping policies for human-in-the-loop robots via deep reinforcement learning","date":"2021-04-26","arxiv_id":"2104.12842","repositories_listed":1,"syntology":null},{"url":"/paper/constraint-guided-reinforcement-learning","slug":"constraint-guided-reinforcement-learning","title":"Constraint-Guided Reinforcement Learning: Augmenting the Agent-Environment-Interaction","date":"2021-04-24","arxiv_id":"2104.11918","repositories_listed":1,"syntology":null},{"url":"/paper/graph-neural-network-reinforcement-learning","slug":"graph-neural-network-reinforcement-learning","title":"Graph Neural Network Reinforcement Learning for Autonomous Mobility-on-Demand Systems","date":"2021-04-23","arxiv_id":"2104.11434","repositories_listed":1,"syntology":null},{"url":"/paper/safe-chance-constrained-reinforcement","slug":"safe-chance-constrained-reinforcement","title":"Safe Chance Constrained Reinforcement Learning for Batch Process Control","date":"2021-04-23","arxiv_id":"2104.11706","repositories_listed":1,"syntology":null},{"url":"/paper/a-learning-gap-between-neuroscience-and","slug":"a-learning-gap-between-neuroscience-and","title":"A learning gap between neuroscience and reinforcement learning","date":"2021-04-22","arxiv_id":"2104.10995","repositories_listed":1,"syntology":null},{"url":"/paper/independent-reinforcement-learning-for-weakly","slug":"independent-reinforcement-learning-for-weakly","title":"Independent Reinforcement Learning for Weakly Cooperative Multiagent Traffic Control Problem","date":"2021-04-22","arxiv_id":"2104.10917","repositories_listed":1,"syntology":null},{"url":"/paper/visual-navigation-with-spatial-attention","slug":"visual-navigation-with-spatial-attention","title":"Visual Navigation with Spatial Attention","date":"2021-04-20","arxiv_id":"2104.09807","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-navigation-with-spatial-attention#ran","syntology_url":"https://syntology.ai/paper/2104.09807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.09807"}},"official":null}},{"url":"/paper/probabilistic-mixture-of-experts-for-1","slug":"probabilistic-mixture-of-experts-for-1","title":"Probabilistic Mixture-of-Experts for Efficient Deep Reinforcement Learning","date":"2021-04-19","arxiv_id":"2104.09122","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/probabilistic-mixture-of-experts-for-1#ran","syntology_url":"https://syntology.ai/paper/2104.09122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.09122"}},"official":{"repos":["JieRen98/rlkit-pmoe"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/keyphrase-generation-with-fine-grained","slug":"keyphrase-generation-with-fine-grained","title":"Keyphrase Generation with Fine-Grained Evaluation-Guided Reinforcement Learning","date":"2021-04-18","arxiv_id":"2104.08799","repositories_listed":1,"syntology":null},{"url":"/paper/quick-learner-automated-vehicle-adapting-its","slug":"quick-learner-automated-vehicle-adapting-its","title":"Quick Learner Automated Vehicle Adapting its Roadmanship to Varying Traffic Cultures with Meta Reinforcement Learning","date":"2021-04-18","arxiv_id":"2104.08876","repositories_listed":1,"syntology":null},{"url":"/paper/learning-on-a-budget-via-teacher-imitation","slug":"learning-on-a-budget-via-teacher-imitation","title":"Learning on a Budget via Teacher Imitation","date":"2021-04-17","arxiv_id":"2104.08440","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-process","slug":"reinforcement-learning-based-process","title":"Reinforcement learning based process optimization and strategy development in conventional tunneling","date":"2021-04-17","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/language-models-are-few-shot-butlers","slug":"language-models-are-few-shot-butlers","title":"Language Models are Few-Shot Butlers","date":"2021-04-16","arxiv_id":"2104.07972","repositories_listed":1,"syntology":null},{"url":"/paper/towards-standardizing-reinforcement-learning","slug":"towards-standardizing-reinforcement-learning","title":"Towards Standardising Reinforcement Learning Approaches for Production Scheduling Problems","date":"2021-04-16","arxiv_id":"2104.08196","repositories_listed":1,"syntology":null},{"url":"/paper/generalising-discrete-action-spaces-with","slug":"generalising-discrete-action-spaces-with","title":"Generalising Discrete Action Spaces with Conditional Action Trees","date":"2021-04-15","arxiv_id":"2104.07294","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/generalising-discrete-action-spaces-with#ran","syntology_url":"https://syntology.ai/paper/2104.07294","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.07294"}},"official":{"repos":["Bam4d/conditional-action-trees"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/quantum-architecture-search-via-deep","slug":"quantum-architecture-search-via-deep","title":"Quantum Architecture Search via Deep Reinforcement Learning","date":"2021-04-15","arxiv_id":"2104.07715","repositories_listed":1,"syntology":null},{"url":"/paper/a-novel-approach-to-curiosity-and-explainable","slug":"a-novel-approach-to-curiosity-and-explainable","title":"A Novel Approach to Curiosity and Explainable Reinforcement Learning via Interpretable Sub-Goals","date":"2021-04-14","arxiv_id":"2104.06630","repositories_listed":1,"syntology":null},{"url":"/paper/decomposed-soft-actor-critic-method-for","slug":"decomposed-soft-actor-critic-method-for","title":"Decomposed Soft Actor-Critic Method for Cooperative Multi-Agent Reinforcement Learning","date":"2021-04-14","arxiv_id":"2104.06655","repositories_listed":1,"syntology":null},{"url":"/paper/safe-continuous-control-with-constrained","slug":"safe-continuous-control-with-constrained","title":"Safe Continuous Control with Constrained Model-Based Policy Optimization","date":"2021-04-14","arxiv_id":"2104.06922","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safe-continuous-control-with-constrained#ran","syntology_url":"https://syntology.ai/paper/2104.06922","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.06922"}},"official":null}},{"url":"/paper/a-coevolutionairy-approach-to-deep-multi","slug":"a-coevolutionairy-approach-to-deep-multi","title":"A coevolutionary approach to deep multi-agent reinforcement learning","date":"2021-04-12","arxiv_id":"2104.05610","repositories_listed":1,"syntology":null},{"url":"/paper/the-atari-data-scraper","slug":"the-atari-data-scraper","title":"The Atari Data Scraper","date":"2021-04-11","arxiv_id":"2104.04893","repositories_listed":1,"syntology":null},{"url":"/paper/imperfect-also-deserves-reward-multi-level","slug":"imperfect-also-deserves-reward-multi-level","title":"Imperfect also Deserves Reward: Multi-Level and Sequential Reward Modeling for Better Dialog Management","date":"2021-04-10","arxiv_id":"2104.04748","repositories_listed":1,"syntology":null},{"url":"/paper/a-bayesian-approach-to-reinforcement-learning","slug":"a-bayesian-approach-to-reinforcement-learning","title":"A Bayesian Approach to Reinforcement Learning of Vision-Based Vehicular Control","date":"2021-04-08","arxiv_id":"2104.03807","repositories_listed":1,"syntology":null},{"url":"/paper/connecting-deep-reinforcement-learning-based","slug":"connecting-deep-reinforcement-learning-based","title":"Connecting Deep-Reinforcement-Learning-based Obstacle Avoidance with Conventional Global Planners using Waypoint Generators","date":"2021-04-08","arxiv_id":"2104.03663","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-time-stepping-for-numerical","slug":"efficient-time-stepping-for-numerical","title":"Efficient time stepping for numerical integration using reinforcement learning","date":"2021-04-08","arxiv_id":"2104.03562","repositories_listed":1,"syntology":null},{"url":"/paper/graph-partitioning-and-sparse-matrix-ordering","slug":"graph-partitioning-and-sparse-matrix-ordering","title":"Graph Partitioning and Sparse Matrix Ordering using Reinforcement Learning and Graph Neural Networks","date":"2021-04-08","arxiv_id":"2104.03546","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-market-making-by-reinforcement","slug":"optimal-market-making-by-reinforcement","title":"Optimal Market Making by Reinforcement Learning","date":"2021-04-08","arxiv_id":"2104.04036","repositories_listed":1,"syntology":null},{"url":"/paper/towards-deployment-of-deep-reinforcement","slug":"towards-deployment-of-deep-reinforcement","title":"Arena-Rosnav: Towards Deployment of Deep-Reinforcement-Learning-Based Obstacle Avoidance into Conventional Autonomous Navigation Systems","date":"2021-04-08","arxiv_id":"2104.03616","repositories_listed":1,"syntology":null}],"record_sha256":"d9803178d1256315678ab12900e41b64a47afba6860fe280272b11e5c2c75001","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}