{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/20","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":20,"pages_in_order":132,"rows_per_page":100,"rows":[1901,2000],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/19","next":"/task/reinforcement-learning/papers/21","papers":[{"url":"/paper/demonstration-free-autonomous-reinforcement","slug":"demonstration-free-autonomous-reinforcement","title":"Demonstration-free Autonomous Reinforcement Learning via Implicit and Bidirectional Curriculum","date":"2023-05-17","arxiv_id":"2305.09943","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":1,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":8,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/demonstration-free-autonomous-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2305.09943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.09943"}},"official":{"repos":["snu-larr/ibc_official"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/explainable-multi-agent-reinforcement","slug":"explainable-multi-agent-reinforcement","title":"Explainable Multi-Agent Reinforcement Learning for Temporal Queries","date":"2023-05-17","arxiv_id":"2305.10378","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-on-google-research","slug":"an-empirical-study-on-google-research","title":"An Empirical Study on Google Research Football Multi-agent Scenarios","date":"2023-05-16","arxiv_id":"2305.09458","repositories_listed":1,"syntology":null},{"url":"/paper/graph-reinforcement-learning-for-network","slug":"graph-reinforcement-learning-for-network","title":"Graph Reinforcement Learning for Network Control via Bi-Level Optimization","date":"2023-05-16","arxiv_id":"2305.09129","repositories_listed":1,"syntology":null},{"url":"/paper/omnisafe-an-infrastructure-for-accelerating","slug":"omnisafe-an-infrastructure-for-accelerating","title":"OmniSafe: An Infrastructure for Accelerating Safe Reinforcement Learning Research","date":"2023-05-16","arxiv_id":"2305.09304","repositories_listed":1,"syntology":null},{"url":"/paper/ramario-experimental-approach-to-reptile","slug":"ramario-experimental-approach-to-reptile","title":"RAMario: Experimental Approach to Reptile Algorithm -- Reinforcement Learning for Mario","date":"2023-05-16","arxiv_id":"2305.09655","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based-exploration","slug":"deep-reinforcement-learning-based-exploration","title":"Deep Reinforcement Learning-based Exploration of Web Applications","date":"2023-05-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/what-matters-in-reinforcement-learning-for","slug":"what-matters-in-reinforcement-learning-for","title":"What Matters in Reinforcement Learning for Tractography","date":"2023-05-15","arxiv_id":"2305.09041","repositories_listed":1,"syntology":null},{"url":"/paper/extracting-diagnosis-pathways-from-electronic","slug":"extracting-diagnosis-pathways-from-electronic","title":"Extracting Diagnosis Pathways from Electronic Health Records Using Deep Reinforcement Learning","date":"2023-05-10","arxiv_id":"2305.06295","repositories_listed":1,"syntology":null},{"url":"/paper/smaclite-a-lightweight-environment-for-multi","slug":"smaclite-a-lightweight-environment-for-multi","title":"SMAClite: A Lightweight Environment for Multi-Agent Reinforcement Learning","date":"2023-05-09","arxiv_id":"2305.05566","repositories_listed":1,"syntology":null},{"url":"/paper/information-design-in-multi-agent","slug":"information-design-in-multi-agent","title":"Information Design in Multi-Agent Reinforcement Learning","date":"2023-05-08","arxiv_id":"2305.06807","repositories_listed":1,"syntology":null},{"url":"/paper/local-optimization-achieves-global-optimality","slug":"local-optimization-achieves-global-optimality","title":"Local Optimization Achieves Global Optimality in Multi-Agent Reinforcement Learning","date":"2023-05-08","arxiv_id":"2305.04819","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/local-optimization-achieves-global-optimality#ran","syntology_url":"https://syntology.ai/paper/2305.04819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.04819"}},"official":{"repos":["zhaoyl18/ratio_game"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-topic-models","slug":"reinforcement-learning-for-topic-models","title":"Reinforcement Learning for Topic Models","date":"2023-05-08","arxiv_id":"2305.04843","repositories_listed":1,"syntology":null},{"url":"/paper/an-asynchronous-updating-reinforcement","slug":"an-asynchronous-updating-reinforcement","title":"An Asynchronous Updating Reinforcement Learning Framework for Task-oriented Dialog System","date":"2023-05-04","arxiv_id":"2305.02718","repositories_listed":1,"syntology":null},{"url":"/paper/explainable-reinforcement-learning-via-a","slug":"explainable-reinforcement-learning-via-a","title":"Explainable Reinforcement Learning via a Causal World Model","date":"2023-05-04","arxiv_id":"2305.02749","repositories_listed":1,"syntology":null},{"url":"/paper/federated-ensemble-directed-offline","slug":"federated-ensemble-directed-offline","title":"Federated Ensemble-Directed Offline Reinforcement Learning","date":"2023-05-04","arxiv_id":"2305.03097","repositories_listed":1,"syntology":null},{"url":"/paper/imap-intrinsically-motivated-adversarial","slug":"imap-intrinsically-motivated-adversarial","title":"Toward Evaluating Robustness of Reinforcement Learning with Adversarial Policy","date":"2023-05-04","arxiv_id":"2305.02605","repositories_listed":1,"syntology":null},{"url":"/paper/simple-noisy-environment-augmentation-for","slug":"simple-noisy-environment-augmentation-for","title":"Simple Noisy Environment Augmentation for Reinforcement Learning","date":"2023-05-04","arxiv_id":"2305.02882","repositories_listed":1,"syntology":null},{"url":"/paper/mixed-integer-optimal-control-via","slug":"mixed-integer-optimal-control-via","title":"Mixed-Integer Optimal Control via Reinforcement Learning: A Case Study on Hybrid Electric Vehicle Energy Management","date":"2023-05-02","arxiv_id":"2305.01461","repositories_listed":1,"syntology":null},{"url":"/paper/catch-collaborative-feature-set-search-for","slug":"catch-collaborative-feature-set-search-for","title":"Catch: Collaborative Feature Set Search for Automated Feature Engineering","date":"2023-04-30","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/posterior-sampling-for-deep-reinforcement","slug":"posterior-sampling-for-deep-reinforcement","title":"Posterior Sampling for Deep Reinforcement Learning","date":"2023-04-30","arxiv_id":"2305.00477","repositories_listed":1,"syntology":null},{"url":"/paper/semi-infinitely-constrained-markov-decision","slug":"semi-infinitely-constrained-markov-decision","title":"Semi-Infinitely Constrained Markov Decision Processes and Efficient Reinforcement Learning","date":"2023-04-29","arxiv_id":"2305.00254","repositories_listed":1,"syntology":null},{"url":"/paper/x-rlflow-graph-reinforcement-learning-for","slug":"x-rlflow-graph-reinforcement-learning-for","title":"X-RLflow: Graph Reinforcement Learning for Neural Network Subgraphs Transformation","date":"2023-04-28","arxiv_id":"2304.14698","repositories_listed":1,"syntology":null},{"url":"/paper/socnavgym-a-reinforcement-learning-gym-for","slug":"socnavgym-a-reinforcement-learning-gym-for","title":"SocNavGym: A Reinforcement Learning Gym for Social Navigation","date":"2023-04-27","arxiv_id":"2304.14102","repositories_listed":1,"syntology":null},{"url":"/paper/quantum-natural-policy-gradients-towards","slug":"quantum-natural-policy-gradients-towards","title":"Quantum Natural Policy Gradients: Towards Sample-Efficient Reinforcement Learning","date":"2023-04-26","arxiv_id":"2304.13571","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quantum-natural-policy-gradients-towards#ran","syntology_url":"https://syntology.ai/paper/2304.13571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.13571"}},"official":{"repos":["nicomeyer96/quantum-natural-policy-gradients"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/partially-observable-mean-field-multi-agent","slug":"partially-observable-mean-field-multi-agent","title":"Partially Observable Mean Field Multi-Agent Reinforcement Learning Based on Graph-Attention","date":"2023-04-25","arxiv_id":"2304.12653","repositories_listed":1,"syntology":null},{"url":"/paper/proto-value-networks-scaling-representation","slug":"proto-value-networks-scaling-representation","title":"Proto-Value Networks: Scaling Representation Learning with Auxiliary Tasks","date":"2023-04-25","arxiv_id":"2304.12567","repositories_listed":1,"syntology":null},{"url":"/paper/proximal-curriculum-for-reinforcement","slug":"proximal-curriculum-for-reinforcement","title":"Proximal Curriculum for Reinforcement Learning Agents","date":"2023-04-25","arxiv_id":"2304.12877","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-approaches-for-traffic","slug":"reinforcement-learning-approaches-for-traffic","title":"Reinforcement Learning Approaches for Traffic Signal Control under Missing Data","date":"2023-04-21","arxiv_id":"2304.10722","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-using-hybrid","slug":"deep-reinforcement-learning-using-hybrid","title":"Deep-Q Learning with Hybrid Quantum Neural Network on Solving Maze Problems","date":"2023-04-20","arxiv_id":"2304.10159","repositories_listed":1,"syntology":null},{"url":"/paper/robust-deep-reinforcement-learning-scheduling","slug":"robust-deep-reinforcement-learning-scheduling","title":"Robust Deep Reinforcement Learning Scheduling via Weight Anchoring","date":"2023-04-20","arxiv_id":"2304.10176","repositories_listed":1,"syntology":null},{"url":"/paper/temporl-laser-pulse-temporal-shape","slug":"temporl-laser-pulse-temporal-shape","title":"TempoRL: laser pulse temporal shape optimization with Deep Reinforcement Learning","date":"2023-04-20","arxiv_id":"2304.12187","repositories_listed":1,"syntology":null},{"url":"/paper/evolving-constrained-reinforcement-learning","slug":"evolving-constrained-reinforcement-learning","title":"Evolving Constrained Reinforcement Learning Policy","date":"2023-04-19","arxiv_id":"2304.09869","repositories_listed":1,"syntology":null},{"url":"/paper/heterogeneous-agent-reinforcement-learning","slug":"heterogeneous-agent-reinforcement-learning","title":"Heterogeneous-Agent Reinforcement Learning","date":"2023-04-19","arxiv_id":"2304.09870","repositories_listed":1,"syntology":null},{"url":"/paper/learning-representative-trajectories-of","slug":"learning-representative-trajectories-of","title":"Learning Representative Trajectories of Dynamical Systems via Domain-Adaptive Imitation","date":"2023-04-19","arxiv_id":"2304.10260","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-model-based-reinforcement","slug":"sample-efficient-model-based-reinforcement","title":"Sample-efficient Model-based Reinforcement Learning for Quantum Control","date":"2023-04-19","arxiv_id":"2304.09718","repositories_listed":1,"syntology":null},{"url":"/paper/stas-spatial-temporal-return-decomposition","slug":"stas-spatial-temporal-return-decomposition","title":"STAS: Spatial-Temporal Return Decomposition for Multi-agent Reinforcement Learning","date":"2023-04-15","arxiv_id":"2304.07520","repositories_listed":1,"syntology":null},{"url":"/paper/language-instructed-reinforcement-learning","slug":"language-instructed-reinforcement-learning","title":"Language Instructed Reinforcement Learning for Human-AI Coordination","date":"2023-04-13","arxiv_id":"2304.07297","repositories_listed":1,"syntology":null},{"url":"/paper/automaton-guided-curriculum-generation-for","slug":"automaton-guided-curriculum-generation-for","title":"Automaton-Guided Curriculum Generation for Reinforcement Learning Agents","date":"2023-04-11","arxiv_id":"2304.05271","repositories_listed":1,"syntology":null},{"url":"/paper/feudal-graph-reinforcement-learning","slug":"feudal-graph-reinforcement-learning","title":"Feudal Graph Reinforcement Learning","date":"2023-04-11","arxiv_id":"2304.05099","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-black-box-model","slug":"reinforcement-learning-based-black-box-model","title":"Reinforcement Learning-Based Black-Box Model Inversion Attacks","date":"2023-04-10","arxiv_id":"2304.04625","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-from-passive-data-via","slug":"reinforcement-learning-from-passive-data-via","title":"Reinforcement Learning from Passive Data via Latent Intentions","date":"2023-04-10","arxiv_id":"2304.04782","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":3,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/reinforcement-learning-from-passive-data-via#ran","syntology_url":"https://syntology.ai/paper/2304.04782","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.04782"}},"official":{"repos":["dibyaghosh/icvf_release"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/robopianist-a-benchmark-for-high-dimensional","slug":"robopianist-a-benchmark-for-high-dimensional","title":"RoboPianist: Dexterous Piano Playing with Deep Reinforcement Learning","date":"2023-04-09","arxiv_id":"2304.04150","repositories_listed":1,"syntology":null},{"url":"/paper/generating-a-graph-colouring-heuristic-with","slug":"generating-a-graph-colouring-heuristic-with","title":"Generating a Graph Colouring Heuristic with Deep Q-Learning and Graph Neural Networks","date":"2023-04-08","arxiv_id":"2304.04051","repositories_listed":1,"syntology":null},{"url":"/paper/uav-obstacle-avoidance-by-human-in-the-loop","slug":"uav-obstacle-avoidance-by-human-in-the-loop","title":"UAV Obstacle Avoidance by Human-in-the-Loop Reinforcement in Arbitrary 3D Environment","date":"2023-04-07","arxiv_id":"2304.05959","repositories_listed":1,"syntology":null},{"url":"/paper/neuroevolution-of-recurrent-architectures-on","slug":"neuroevolution-of-recurrent-architectures-on","title":"Neuroevolution of Recurrent Architectures on Control Tasks","date":"2023-04-03","arxiv_id":"2304.12431","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-goal-reaching-reinforcement-learning","slug":"optimal-goal-reaching-reinforcement-learning","title":"Optimal Goal-Reaching Reinforcement Learning via Quasimetric Learning","date":"2023-04-03","arxiv_id":"2304.01203","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimal-goal-reaching-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2304.01203","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.01203"}},"official":{"repos":["quasimetric-learning/quasimetric-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/swarm-reinforcement-learning-for-adaptive-1","slug":"swarm-reinforcement-learning-for-adaptive-1","title":"Swarm Reinforcement Learning For Adaptive Mesh Refinement","date":"2023-04-03","arxiv_id":"2304.00818","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/swarm-reinforcement-learning-for-adaptive-1#ran","syntology_url":"https://syntology.ai/paper/2304.00818","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.00818"}},"official":{"repos":["niklasfreymuth/asmr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-context-distribution-shift-in-task","slug":"on-context-distribution-shift-in-task","title":"On Context Distribution Shift in Task Representation Learning for Offline Meta RL","date":"2023-04-01","arxiv_id":"2304.00354","repositories_listed":1,"syntology":null},{"url":"/paper/solving-dynamic-traveling-salesman-problems","slug":"solving-dynamic-traveling-salesman-problems","title":"Solving Dynamic Traveling Salesman Problems With Deep Reinforcement Learning","date":"2023-04-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mahalo-unifying-offline-reinforcement","slug":"mahalo-unifying-offline-reinforcement","title":"MAHALO: Unifying Offline Reinforcement Learning and Imitation Learning from Observations","date":"2023-03-30","arxiv_id":"2303.17156","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mahalo-unifying-offline-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2303.17156","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17156"}},"official":{"repos":["anqili/mahalo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/utilizing-reinforcement-learning-for-de-novo","slug":"utilizing-reinforcement-learning-for-de-novo","title":"Utilizing Reinforcement Learning for de novo Drug Design","date":"2023-03-30","arxiv_id":"2303.17615","repositories_listed":1,"syntology":null},{"url":"/paper/pgx-hardware-accelerated-parallel-game-1","slug":"pgx-hardware-accelerated-parallel-game-1","title":"Pgx: Hardware-Accelerated Parallel Game Simulators for Reinforcement Learning","date":"2023-03-29","arxiv_id":"2303.17503","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/pgx-hardware-accelerated-parallel-game-1#ran","syntology_url":"https://syntology.ai/paper/2303.17503","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17503"}},"official":{"repos":["sotetsuk/pgx"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/model-based-reinforcement-learning-with-3","slug":"model-based-reinforcement-learning-with-3","title":"Model-Based Reinforcement Learning with Isolated Imaginations","date":"2023-03-27","arxiv_id":"2303.14889","repositories_listed":1,"syntology":null},{"url":"/paper/inverse-reinforcement-learning-without","slug":"inverse-reinforcement-learning-without","title":"Inverse Reinforcement Learning without Reinforcement Learning","date":"2023-03-26","arxiv_id":"2303.14623","repositories_listed":1,"syntology":null},{"url":"/paper/marl-jax-multi-agent-reinforcement-leaning","slug":"marl-jax-multi-agent-reinforcement-leaning","title":"marl-jax: Multi-Agent Reinforcement Leaning Framework","date":"2023-03-24","arxiv_id":"2303.13808","repositories_listed":1,"syntology":null},{"url":"/paper/safe-and-sample-efficient-reinforcement","slug":"safe-and-sample-efficient-reinforcement","title":"Safe and Sample-efficient Reinforcement Learning for Clustered Dynamic Environments","date":"2023-03-24","arxiv_id":"2303.14265","repositories_listed":1,"syntology":null},{"url":"/paper/rlor-a-flexible-framework-of-deep","slug":"rlor-a-flexible-framework-of-deep","title":"RLOR: A Flexible Framework of Deep Reinforcement Learning for Operation Research","date":"2023-03-23","arxiv_id":"2303.13117","repositories_listed":1,"syntology":null},{"url":"/paper/clip4mc-an-rl-friendly-vision-language-model","slug":"clip4mc-an-rl-friendly-vision-language-model","title":"Reinforcement Learning Friendly Vision-Language Model for Minecraft","date":"2023-03-19","arxiv_id":"2303.10571","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clip4mc-an-rl-friendly-vision-language-model#ran","syntology_url":"https://syntology.ai/paper/2303.10571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.10571"}},"official":{"repos":["PKU-RL/CLIP4MC"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-update-to-data-ratio-minimizing-world","slug":"dynamic-update-to-data-ratio-minimizing-world","title":"Dynamic Update-to-Data Ratio: Minimizing World Model Overfitting","date":"2023-03-17","arxiv_id":"2303.10144","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-update-to-data-ratio-minimizing-world#ran","syntology_url":"https://syntology.ai/paper/2303.10144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.10144"}},"official":{"repos":["nicolinho/dutd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/conditionally-optimistic-exploration-for","slug":"conditionally-optimistic-exploration-for","title":"Conditionally Optimistic Exploration for Cooperative Deep Multi-Agent Reinforcement Learning","date":"2023-03-16","arxiv_id":"2303.09032","repositories_listed":1,"syntology":null},{"url":"/paper/decentralized-multi-agent-reinforcement-4","slug":"decentralized-multi-agent-reinforcement-4","title":"Decentralized Multi-Agent Reinforcement Learning for Continuous-Space Stochastic Games","date":"2023-03-16","arxiv_id":"2303.13539","repositories_listed":1,"syntology":null},{"url":"/paper/kernel-density-bayesian-inverse-reinforcement","slug":"kernel-density-bayesian-inverse-reinforcement","title":"Kernel Density Bayesian Inverse Reinforcement Learning","date":"2023-03-13","arxiv_id":"2303.06827","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/kernel-density-bayesian-inverse-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2303.06827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06827"}},"official":{"repos":["bee-hive/kdbirl_public"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/transformer-based-world-models-are-happy-with","slug":"transformer-based-world-models-are-happy-with","title":"Transformer-based World Models Are Happy With 100k Interactions","date":"2023-03-13","arxiv_id":"2303.07109","repositories_listed":1,"syntology":{"n":25,"n_ran":16,"n_constructed":6,"n_ran_checked":8,"n_instrument":8,"n_unverified":9,"n_honours":1,"n_violates":1,"n_no_contract":6,"n_pointer_only":0,"phrase":"16 ran (of which 6 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 8 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/transformer-based-world-models-are-happy-with#ran","syntology_url":"https://syntology.ai/paper/2303.07109","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.07109"}},"official":{"repos":["jrobine/twm"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":6,"n_ran_no_instrument_failure":8,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/optimal-foraging-strategies-can-be-learned","slug":"optimal-foraging-strategies-can-be-learned","title":"Optimal foraging strategies can be learned","date":"2023-03-10","arxiv_id":"2303.06050","repositories_listed":1,"syntology":null},{"url":"/paper/raccer-towards-reachable-and-certain","slug":"raccer-towards-reachable-and-certain","title":"RACCER: Towards Reachable and Certain Counterfactual Explanations for Reinforcement Learning","date":"2023-03-08","arxiv_id":"2303.04475","repositories_listed":1,"syntology":null},{"url":"/paper/a-multiplicative-value-function-for-safe-and","slug":"a-multiplicative-value-function-for-safe-and","title":"A Multiplicative Value Function for Safe and Efficient Reinforcement Learning","date":"2023-03-07","arxiv_id":"2303.04118","repositories_listed":1,"syntology":null},{"url":"/paper/learning-when-to-treat-business-processes","slug":"learning-when-to-treat-business-processes","title":"Learning When to Treat Business Processes: Prescriptive Process Monitoring with Causal Inference and Reinforcement Learning","date":"2023-03-07","arxiv_id":"2303.03572","repositories_listed":1,"syntology":null},{"url":"/paper/zeroth-order-optimization-meets-human","slug":"zeroth-order-optimization-meets-human","title":"Zeroth-Order Optimization Meets Human Feedback: Provable Learning via Ranking Oracles","date":"2023-03-07","arxiv_id":"2303.03751","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/zeroth-order-optimization-meets-human#ran","syntology_url":"https://syntology.ai/paper/2303.03751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03751"}},"official":{"repos":["TZW1998/Taming-Stable-Diffusion-with-Human-Ranking-Feedback"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-reinforcement-learning-via-probabilistic","slug":"safe-reinforcement-learning-via-probabilistic","title":"Safe Reinforcement Learning via Probabilistic Logic Shields","date":"2023-03-06","arxiv_id":"2303.03226","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/safe-reinforcement-learning-via-probabilistic#ran","syntology_url":"https://syntology.ai/paper/2303.03226","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03226"}},"official":null}},{"url":"/paper/bounding-the-optimal-value-function-in","slug":"bounding-the-optimal-value-function-in","title":"Bounding the Optimal Value Function in Compositional Reinforcement Learning","date":"2023-03-05","arxiv_id":"2303.02557","repositories_listed":1,"syntology":null},{"url":"/paper/improved-sample-complexity-bounds-for","slug":"improved-sample-complexity-bounds-for","title":"Improved Sample Complexity Bounds for Distributionally Robust Reinforcement Learning","date":"2023-03-05","arxiv_id":"2303.02783","repositories_listed":1,"syntology":null},{"url":"/paper/swim-a-general-purpose-high-performing-and","slug":"swim-a-general-purpose-high-performing-and","title":"Swim: A General-Purpose, High-Performing, and Efficient Activation Function for Locomotion Control Tasks","date":"2023-03-05","arxiv_id":"2303.02640","repositories_listed":1,"syntology":null},{"url":"/paper/corl-environment-creation-and-management","slug":"corl-environment-creation-and-management","title":"CoRL: Environment Creation and Management Focused on System Integration","date":"2023-03-03","arxiv_id":"2303.02182","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-guided-multi-objective","slug":"reinforcement-learning-guided-multi-objective","title":"Reinforcement Learning Guided Multi-Objective Exam Paper Generation","date":"2023-03-02","arxiv_id":"2303.01042","repositories_listed":1,"syntology":null},{"url":"/paper/ls-iq-implicit-reward-regularization-for","slug":"ls-iq-implicit-reward-regularization-for","title":"LS-IQ: Implicit Reward Regularization for Inverse Reinforcement Learning","date":"2023-03-01","arxiv_id":"2303.00599","repositories_listed":1,"syntology":null},{"url":"/paper/human-inspired-framework-to-accelerate","slug":"human-inspired-framework-to-accelerate","title":"Human-Inspired Framework to Accelerate Reinforcement Learning","date":"2023-02-28","arxiv_id":"2303.08115","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-uncertainty-in-value-functions","slug":"model-based-uncertainty-in-value-functions","title":"Model-Based Uncertainty in Value Functions","date":"2023-02-24","arxiv_id":"2302.12526","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/model-based-uncertainty-in-value-functions#ran","syntology_url":"https://syntology.ai/paper/2302.12526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12526"}},"official":{"repos":["boschresearch/ube-mbrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/assessment-of-reinforcement-learning-for","slug":"assessment-of-reinforcement-learning-for","title":"Assessment of Reinforcement Learning for Macro Placement","date":"2023-02-21","arxiv_id":"2302.11014","repositories_listed":1,"syntology":null},{"url":"/paper/mac-po-multi-agent-experience-replay-via","slug":"mac-po-multi-agent-experience-replay-via","title":"MAC-PO: Multi-Agent Experience Replay via Collective Priority Optimization","date":"2023-02-21","arxiv_id":"2302.10418","repositories_listed":1,"syntology":null},{"url":"/paper/minimax-bayes-reinforcement-learning","slug":"minimax-bayes-reinforcement-learning","title":"Minimax-Bayes Reinforcement Learning","date":"2023-02-21","arxiv_id":"2302.10831","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-cost","slug":"deep-reinforcement-learning-for-cost","title":"Deep Reinforcement Learning for Cost-Effective Medical Diagnosis","date":"2023-02-20","arxiv_id":"2302.10261","repositories_listed":1,"syntology":null},{"url":"/paper/multiagent-inverse-reinforcement-learning-via","slug":"multiagent-inverse-reinforcement-learning-via","title":"Multiagent Inverse Reinforcement Learning via Theory of Mind Reasoning","date":"2023-02-20","arxiv_id":"2302.10238","repositories_listed":1,"syntology":null},{"url":"/paper/generalization-in-visual-reinforcement","slug":"generalization-in-visual-reinforcement","title":"Generalization in Visual Reinforcement Learning with the Reward Sequence Distribution","date":"2023-02-19","arxiv_id":"2302.09601","repositories_listed":1,"syntology":null},{"url":"/paper/post-episodic-reinforcement-learning","slug":"post-episodic-reinforcement-learning","title":"Post Reinforcement Learning Inference","date":"2023-02-17","arxiv_id":"2302.08854","repositories_listed":1,"syntology":null},{"url":"/paper/swapped-goal-conditioned-offline","slug":"swapped-goal-conditioned-offline","title":"Swapped goal-conditioned offline reinforcement learning","date":"2023-02-17","arxiv_id":"2302.08865","repositories_listed":1,"syntology":null},{"url":"/paper/conservative-state-value-estimation-for-1","slug":"conservative-state-value-estimation-for-1","title":"Conservative State Value Estimation for Offline Reinforcement Learning","date":"2023-02-14","arxiv_id":"2302.06884","repositories_listed":1,"syntology":null},{"url":"/paper/constrained-decision-transformer-for-offline","slug":"constrained-decision-transformer-for-offline","title":"Constrained Decision Transformer for Offline Safe Reinforcement Learning","date":"2023-02-14","arxiv_id":"2302.07351","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/constrained-decision-transformer-for-offline#ran","syntology_url":"https://syntology.ai/paper/2302.07351","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.07351"}},"official":{"repos":["liuzuxin/osrl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-deep-reinforcement-learning-through-2","slug":"robust-deep-reinforcement-learning-through-2","title":"Regret-Based Defense in Adversarial Reinforcement Learning","date":"2023-02-14","arxiv_id":"2302.06912","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robust-deep-reinforcement-learning-through-2#ran","syntology_url":"https://syntology.ai/paper/2302.06912","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.06912"}},"official":{"repos":["romanbelaire/robust-ccer"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/semiconductor-fab-scheduling-with-self","slug":"semiconductor-fab-scheduling-with-self","title":"Semiconductor Fab Scheduling with Self-Supervised and Reinforcement Learning","date":"2023-02-14","arxiv_id":"2302.07162","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-noise-filtering-with-dynamic-sparse","slug":"automatic-noise-filtering-with-dynamic-sparse","title":"Automatic Noise Filtering with Dynamic Sparse Training in Deep Reinforcement Learning","date":"2023-02-13","arxiv_id":"2302.06548","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/automatic-noise-filtering-with-dynamic-sparse#ran","syntology_url":"https://syntology.ai/paper/2302.06548","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.06548"}},"official":{"repos":["bramgrooten/automatic-noise-filtering"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/guiding-pretraining-in-reinforcement-learning","slug":"guiding-pretraining-in-reinforcement-learning","title":"Guiding Pretraining in Reinforcement Learning with Large Language Models","date":"2023-02-13","arxiv_id":"2302.06692","repositories_listed":1,"syntology":null},{"url":"/paper/a-large-parametrized-space-of-meta","slug":"a-large-parametrized-space-of-meta","title":"Procedural generation of meta-reinforcement learning tasks","date":"2023-02-11","arxiv_id":"2302.05583","repositories_listed":1,"syntology":null},{"url":"/paper/a-swat-based-reinforcement-learning-framework","slug":"a-swat-based-reinforcement-learning-framework","title":"A SWAT-based Reinforcement Learning Framework for Crop Management","date":"2023-02-10","arxiv_id":"2302.04988","repositories_listed":1,"syntology":null},{"url":"/paper/robust-knowledge-transfer-in-tiered-1","slug":"robust-knowledge-transfer-in-tiered-1","title":"Robust Knowledge Transfer in Tiered Reinforcement Learning","date":"2023-02-10","arxiv_id":"2302.05534","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-knowledge-transfer-in-tiered-1#ran","syntology_url":"https://syntology.ai/paper/2302.05534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.05534"}},"official":{"repos":["jiaweihhuang/robust-tiered-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-wisdom-of-hindsight-makes-language-models","slug":"the-wisdom-of-hindsight-makes-language-models","title":"The Wisdom of Hindsight Makes Language Models Better Instruction Followers","date":"2023-02-10","arxiv_id":"2302.05206","repositories_listed":1,"syntology":null},{"url":"/paper/attacking-cooperative-multi-agent","slug":"attacking-cooperative-multi-agent","title":"Attacking Cooperative Multi-Agent Reinforcement Learning by Adversarial Minority Influence","date":"2023-02-07","arxiv_id":"2302.03322","repositories_listed":1,"syntology":null},{"url":"/paper/multi-task-recommendations-with-reinforcement","slug":"multi-task-recommendations-with-reinforcement","title":"Multi-Task Recommendations with Reinforcement Learning","date":"2023-02-07","arxiv_id":"2302.03328","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-task-recommendations-with-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2302.03328","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.03328"}},"official":{"repos":["applied-machine-learning-lab/rmtl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-free-quantum-gate-design-and","slug":"model-free-quantum-gate-design-and","title":"Model-free Quantum Gate Design and Calibration using Deep Reinforcement Learning","date":"2023-02-05","arxiv_id":"2302.02371","repositories_listed":1,"syntology":null},{"url":"/paper/distributional-constrained-reinforcement","slug":"distributional-constrained-reinforcement","title":"Distributional constrained reinforcement learning for supply chain optimization","date":"2023-02-03","arxiv_id":"2302.01727","repositories_listed":1,"syntology":null}],"record_sha256":"e7df516d19336e6d2c236f2ceea9a783a317a4abc31385c5df1b823c5e8a7268","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}