{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/20","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":20,"pages_in_order":135,"rows_per_page":100,"rows":[1901,2000],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/19","next":"/task/reinforcement-learning-2/papers/21","papers":[{"url":"/paper/deep-reinforcement-learning-using-hybrid","slug":"deep-reinforcement-learning-using-hybrid","title":"Deep-Q Learning with Hybrid Quantum Neural Network on Solving Maze Problems","date":"2023-04-20","arxiv_id":"2304.10159","repositories_listed":1,"syntology":null},{"url":"/paper/robust-deep-reinforcement-learning-scheduling","slug":"robust-deep-reinforcement-learning-scheduling","title":"Robust Deep Reinforcement Learning Scheduling via Weight Anchoring","date":"2023-04-20","arxiv_id":"2304.10176","repositories_listed":1,"syntology":null},{"url":"/paper/temporl-laser-pulse-temporal-shape","slug":"temporl-laser-pulse-temporal-shape","title":"TempoRL: laser pulse temporal shape optimization with Deep Reinforcement Learning","date":"2023-04-20","arxiv_id":"2304.12187","repositories_listed":1,"syntology":null},{"url":"/paper/evolving-constrained-reinforcement-learning","slug":"evolving-constrained-reinforcement-learning","title":"Evolving Constrained Reinforcement Learning Policy","date":"2023-04-19","arxiv_id":"2304.09869","repositories_listed":1,"syntology":null},{"url":"/paper/h-tsp-hierarchically-solving-the-large-scale","slug":"h-tsp-hierarchically-solving-the-large-scale","title":"H-TSP: Hierarchically Solving the Large-Scale Travelling Salesman Problem","date":"2023-04-19","arxiv_id":"2304.09395","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/h-tsp-hierarchically-solving-the-large-scale#ran","syntology_url":"https://syntology.ai/paper/2304.09395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.09395"}},"official":{"repos":["Learning4Optimization-HUST/H-TSP"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/heterogeneous-agent-reinforcement-learning","slug":"heterogeneous-agent-reinforcement-learning","title":"Heterogeneous-Agent Reinforcement Learning","date":"2023-04-19","arxiv_id":"2304.09870","repositories_listed":1,"syntology":null},{"url":"/paper/learning-representative-trajectories-of","slug":"learning-representative-trajectories-of","title":"Learning Representative Trajectories of Dynamical Systems via Domain-Adaptive Imitation","date":"2023-04-19","arxiv_id":"2304.10260","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-model-based-reinforcement","slug":"sample-efficient-model-based-reinforcement","title":"Sample-efficient Model-based Reinforcement Learning for Quantum Control","date":"2023-04-19","arxiv_id":"2304.09718","repositories_listed":1,"syntology":null},{"url":"/paper/using-offline-data-to-speed-up-reinforcement","slug":"using-offline-data-to-speed-up-reinforcement","title":"Using Offline Data to Speed Up Reinforcement Learning in Procedurally Generated Environments","date":"2023-04-18","arxiv_id":"2304.09825","repositories_listed":1,"syntology":null},{"url":"/paper/stas-spatial-temporal-return-decomposition","slug":"stas-spatial-temporal-return-decomposition","title":"STAS: Spatial-Temporal Return Decomposition for Multi-agent Reinforcement Learning","date":"2023-04-15","arxiv_id":"2304.07520","repositories_listed":1,"syntology":null},{"url":"/paper/language-instructed-reinforcement-learning","slug":"language-instructed-reinforcement-learning","title":"Language Instructed Reinforcement Learning for Human-AI Coordination","date":"2023-04-13","arxiv_id":"2304.07297","repositories_listed":1,"syntology":null},{"url":"/paper/automaton-guided-curriculum-generation-for","slug":"automaton-guided-curriculum-generation-for","title":"Automaton-Guided Curriculum Generation for Reinforcement Learning Agents","date":"2023-04-11","arxiv_id":"2304.05271","repositories_listed":1,"syntology":null},{"url":"/paper/feudal-graph-reinforcement-learning","slug":"feudal-graph-reinforcement-learning","title":"Feudal Graph Reinforcement Learning","date":"2023-04-11","arxiv_id":"2304.05099","repositories_listed":1,"syntology":null},{"url":"/paper/eagle-end-to-end-deep-reinforcement-learning","slug":"eagle-end-to-end-deep-reinforcement-learning","title":"Eagle: End-to-end Deep Reinforcement Learning based Autonomous Control of PTZ Cameras","date":"2023-04-10","arxiv_id":"2304.04356","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-black-box-model","slug":"reinforcement-learning-based-black-box-model","title":"Reinforcement Learning-Based Black-Box Model Inversion Attacks","date":"2023-04-10","arxiv_id":"2304.04625","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-from-passive-data-via","slug":"reinforcement-learning-from-passive-data-via","title":"Reinforcement Learning from Passive Data via Latent Intentions","date":"2023-04-10","arxiv_id":"2304.04782","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":3,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/reinforcement-learning-from-passive-data-via#ran","syntology_url":"https://syntology.ai/paper/2304.04782","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.04782"}},"official":{"repos":["dibyaghosh/icvf_release"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/respect-reinforcement-learning-based-edge","slug":"respect-reinforcement-learning-based-edge","title":"RESPECT: Reinforcement Learning based Edge Scheduling on Pipelined Coral Edge TPUs","date":"2023-04-10","arxiv_id":"2304.04716","repositories_listed":1,"syntology":null},{"url":"/paper/uncertainty-driven-trajectory-truncation-for","slug":"uncertainty-driven-trajectory-truncation-for","title":"Uncertainty-driven Trajectory Truncation for Data Augmentation in Offline Reinforcement Learning","date":"2023-04-10","arxiv_id":"2304.04660","repositories_listed":1,"syntology":null},{"url":"/paper/robopianist-a-benchmark-for-high-dimensional","slug":"robopianist-a-benchmark-for-high-dimensional","title":"RoboPianist: Dexterous Piano Playing with Deep Reinforcement Learning","date":"2023-04-09","arxiv_id":"2304.04150","repositories_listed":1,"syntology":null},{"url":"/paper/generating-a-graph-colouring-heuristic-with","slug":"generating-a-graph-colouring-heuristic-with","title":"Generating a Graph Colouring Heuristic with Deep Q-Learning and Graph Neural Networks","date":"2023-04-08","arxiv_id":"2304.04051","repositories_listed":1,"syntology":null},{"url":"/paper/marl-idr-multi-agent-reinforcement-learning","slug":"marl-idr-multi-agent-reinforcement-learning","title":"MARL-iDR: Multi-Agent Reinforcement Learning for Incentive-based Residential Demand Response","date":"2023-04-08","arxiv_id":"2304.04086","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based-mapless","slug":"deep-reinforcement-learning-based-mapless","title":"Deep Reinforcement Learning-Based Mapless Crowd Navigation with Perceived Risk of the Moving Crowd for Mobile Robots","date":"2023-04-07","arxiv_id":"2304.03593","repositories_listed":1,"syntology":null},{"url":"/paper/uav-obstacle-avoidance-by-human-in-the-loop","slug":"uav-obstacle-avoidance-by-human-in-the-loop","title":"UAV Obstacle Avoidance by Human-in-the-Loop Reinforcement in Arbitrary 3D Environment","date":"2023-04-07","arxiv_id":"2304.05959","repositories_listed":1,"syntology":null},{"url":"/paper/neuroevolution-of-recurrent-architectures-on","slug":"neuroevolution-of-recurrent-architectures-on","title":"Neuroevolution of Recurrent Architectures on Control Tasks","date":"2023-04-03","arxiv_id":"2304.12431","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-goal-reaching-reinforcement-learning","slug":"optimal-goal-reaching-reinforcement-learning","title":"Optimal Goal-Reaching Reinforcement Learning via Quasimetric Learning","date":"2023-04-03","arxiv_id":"2304.01203","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimal-goal-reaching-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2304.01203","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.01203"}},"official":{"repos":["quasimetric-learning/quasimetric-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/swarm-reinforcement-learning-for-adaptive-1","slug":"swarm-reinforcement-learning-for-adaptive-1","title":"Swarm Reinforcement Learning For Adaptive Mesh Refinement","date":"2023-04-03","arxiv_id":"2304.00818","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/swarm-reinforcement-learning-for-adaptive-1#ran","syntology_url":"https://syntology.ai/paper/2304.00818","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.00818"}},"official":{"repos":["niklasfreymuth/asmr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-context-distribution-shift-in-task","slug":"on-context-distribution-shift-in-task","title":"On Context Distribution Shift in Task Representation Learning for Offline Meta RL","date":"2023-04-01","arxiv_id":"2304.00354","repositories_listed":1,"syntology":null},{"url":"/paper/recover-triggered-states-protect-model","slug":"recover-triggered-states-protect-model","title":"Recover Triggered States: Protect Model Against Backdoor Attack in Reinforcement Learning","date":"2023-04-01","arxiv_id":"2304.00252","repositories_listed":1,"syntology":null},{"url":"/paper/solving-dynamic-traveling-salesman-problems","slug":"solving-dynamic-traveling-salesman-problems","title":"Solving Dynamic Traveling Salesman Problems With Deep Reinforcement Learning","date":"2023-04-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mahalo-unifying-offline-reinforcement","slug":"mahalo-unifying-offline-reinforcement","title":"MAHALO: Unifying Offline Reinforcement Learning and Imitation Learning from Observations","date":"2023-03-30","arxiv_id":"2303.17156","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mahalo-unifying-offline-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2303.17156","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17156"}},"official":{"repos":["anqili/mahalo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/utilizing-reinforcement-learning-for-de-novo","slug":"utilizing-reinforcement-learning-for-de-novo","title":"Utilizing Reinforcement Learning for de novo Drug Design","date":"2023-03-30","arxiv_id":"2303.17615","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-with-6","slug":"multi-agent-reinforcement-learning-with-6","title":"Multi-Agent Reinforcement Learning with Action Masking for UAV-enabled Mobile Communications","date":"2023-03-29","arxiv_id":"2303.16737","repositories_listed":1,"syntology":null},{"url":"/paper/pgx-hardware-accelerated-parallel-game-1","slug":"pgx-hardware-accelerated-parallel-game-1","title":"Pgx: Hardware-Accelerated Parallel Game Simulators for Reinforcement Learning","date":"2023-03-29","arxiv_id":"2303.17503","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/pgx-hardware-accelerated-parallel-game-1#ran","syntology_url":"https://syntology.ai/paper/2303.17503","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17503"}},"official":{"repos":["sotetsuk/pgx"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/model-based-reinforcement-learning-with-3","slug":"model-based-reinforcement-learning-with-3","title":"Model-Based Reinforcement Learning with Isolated Imaginations","date":"2023-03-27","arxiv_id":"2303.14889","repositories_listed":1,"syntology":null},{"url":"/paper/balancing-policy-constraint-and-ensemble-size","slug":"balancing-policy-constraint-and-ensemble-size","title":"Balancing policy constraint and ensemble size in uncertainty-based offline reinforcement learning","date":"2023-03-26","arxiv_id":"2303.14716","repositories_listed":1,"syntology":null},{"url":"/paper/inverse-reinforcement-learning-without","slug":"inverse-reinforcement-learning-without","title":"Inverse Reinforcement Learning without Reinforcement Learning","date":"2023-03-26","arxiv_id":"2303.14623","repositories_listed":1,"syntology":null},{"url":"/paper/marl-jax-multi-agent-reinforcement-leaning","slug":"marl-jax-multi-agent-reinforcement-leaning","title":"marl-jax: Multi-Agent Reinforcement Leaning Framework","date":"2023-03-24","arxiv_id":"2303.13808","repositories_listed":1,"syntology":null},{"url":"/paper/safe-and-sample-efficient-reinforcement","slug":"safe-and-sample-efficient-reinforcement","title":"Safe and Sample-efficient Reinforcement Learning for Clustered Dynamic Environments","date":"2023-03-24","arxiv_id":"2303.14265","repositories_listed":1,"syntology":null},{"url":"/paper/rlor-a-flexible-framework-of-deep","slug":"rlor-a-flexible-framework-of-deep","title":"RLOR: A Flexible Framework of Deep Reinforcement Learning for Operation Research","date":"2023-03-23","arxiv_id":"2303.13117","repositories_listed":1,"syntology":null},{"url":"/paper/clip4mc-an-rl-friendly-vision-language-model","slug":"clip4mc-an-rl-friendly-vision-language-model","title":"Reinforcement Learning Friendly Vision-Language Model for Minecraft","date":"2023-03-19","arxiv_id":"2303.10571","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clip4mc-an-rl-friendly-vision-language-model#ran","syntology_url":"https://syntology.ai/paper/2303.10571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.10571"}},"official":{"repos":["PKU-RL/CLIP4MC"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-update-to-data-ratio-minimizing-world","slug":"dynamic-update-to-data-ratio-minimizing-world","title":"Dynamic Update-to-Data Ratio: Minimizing World Model Overfitting","date":"2023-03-17","arxiv_id":"2303.10144","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-update-to-data-ratio-minimizing-world#ran","syntology_url":"https://syntology.ai/paper/2303.10144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.10144"}},"official":{"repos":["nicolinho/dutd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/conditionally-optimistic-exploration-for","slug":"conditionally-optimistic-exploration-for","title":"Conditionally Optimistic Exploration for Cooperative Deep Multi-Agent Reinforcement Learning","date":"2023-03-16","arxiv_id":"2303.09032","repositories_listed":1,"syntology":null},{"url":"/paper/decentralized-multi-agent-reinforcement-4","slug":"decentralized-multi-agent-reinforcement-4","title":"Decentralized Multi-Agent Reinforcement Learning for Continuous-Space Stochastic Games","date":"2023-03-16","arxiv_id":"2303.13539","repositories_listed":1,"syntology":null},{"url":"/paper/act-then-measure-reinforcement-learning-for","slug":"act-then-measure-reinforcement-learning-for","title":"Act-Then-Measure: Reinforcement Learning for Partially Observable Environments with Active Measuring","date":"2023-03-14","arxiv_id":"2303.08271","repositories_listed":1,"syntology":null},{"url":"/paper/kernel-density-bayesian-inverse-reinforcement","slug":"kernel-density-bayesian-inverse-reinforcement","title":"Kernel Density Bayesian Inverse Reinforcement Learning","date":"2023-03-13","arxiv_id":"2303.06827","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/kernel-density-bayesian-inverse-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2303.06827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06827"}},"official":{"repos":["bee-hive/kdbirl_public"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/transformer-based-world-models-are-happy-with","slug":"transformer-based-world-models-are-happy-with","title":"Transformer-based World Models Are Happy With 100k Interactions","date":"2023-03-13","arxiv_id":"2303.07109","repositories_listed":1,"syntology":{"n":25,"n_ran":16,"n_constructed":6,"n_ran_checked":8,"n_instrument":8,"n_unverified":9,"n_honours":1,"n_violates":1,"n_no_contract":6,"n_pointer_only":0,"phrase":"16 ran (of which 6 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 8 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/transformer-based-world-models-are-happy-with#ran","syntology_url":"https://syntology.ai/paper/2303.07109","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.07109"}},"official":{"repos":["jrobine/twm"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":6,"n_ran_no_instrument_failure":8,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/optimal-foraging-strategies-can-be-learned","slug":"optimal-foraging-strategies-can-be-learned","title":"Optimal foraging strategies can be learned","date":"2023-03-10","arxiv_id":"2303.06050","repositories_listed":1,"syntology":null},{"url":"/paper/raccer-towards-reachable-and-certain","slug":"raccer-towards-reachable-and-certain","title":"RACCER: Towards Reachable and Certain Counterfactual Explanations for Reinforcement Learning","date":"2023-03-08","arxiv_id":"2303.04475","repositories_listed":1,"syntology":null},{"url":"/paper/a-multiplicative-value-function-for-safe-and","slug":"a-multiplicative-value-function-for-safe-and","title":"A Multiplicative Value Function for Safe and Efficient Reinforcement Learning","date":"2023-03-07","arxiv_id":"2303.04118","repositories_listed":1,"syntology":null},{"url":"/paper/diminishing-return-of-value-expansion-methods","slug":"diminishing-return-of-value-expansion-methods","title":"Diminishing Return of Value Expansion Methods in Model-Based Reinforcement Learning","date":"2023-03-07","arxiv_id":"2303.03955","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/diminishing-return-of-value-expansion-methods#ran","syntology_url":"https://syntology.ai/paper/2303.03955","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03955"}},"official":{"repos":["danielpalen/value_expansion"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-when-to-treat-business-processes","slug":"learning-when-to-treat-business-processes","title":"Learning When to Treat Business Processes: Prescriptive Process Monitoring with Causal Inference and Reinforcement Learning","date":"2023-03-07","arxiv_id":"2303.03572","repositories_listed":1,"syntology":null},{"url":"/paper/zeroth-order-optimization-meets-human","slug":"zeroth-order-optimization-meets-human","title":"Zeroth-Order Optimization Meets Human Feedback: Provable Learning via Ranking Oracles","date":"2023-03-07","arxiv_id":"2303.03751","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/zeroth-order-optimization-meets-human#ran","syntology_url":"https://syntology.ai/paper/2303.03751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03751"}},"official":{"repos":["TZW1998/Taming-Stable-Diffusion-with-Human-Ranking-Feedback"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-reinforcement-learning-via-probabilistic","slug":"safe-reinforcement-learning-via-probabilistic","title":"Safe Reinforcement Learning via Probabilistic Logic Shields","date":"2023-03-06","arxiv_id":"2303.03226","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/safe-reinforcement-learning-via-probabilistic#ran","syntology_url":"https://syntology.ai/paper/2303.03226","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03226"}},"official":null}},{"url":"/paper/bounding-the-optimal-value-function-in","slug":"bounding-the-optimal-value-function-in","title":"Bounding the Optimal Value Function in Compositional Reinforcement Learning","date":"2023-03-05","arxiv_id":"2303.02557","repositories_listed":1,"syntology":null},{"url":"/paper/improved-sample-complexity-bounds-for","slug":"improved-sample-complexity-bounds-for","title":"Improved Sample Complexity Bounds for Distributionally Robust Reinforcement Learning","date":"2023-03-05","arxiv_id":"2303.02783","repositories_listed":1,"syntology":null},{"url":"/paper/swim-a-general-purpose-high-performing-and","slug":"swim-a-general-purpose-high-performing-and","title":"Swim: A General-Purpose, High-Performing, and Efficient Activation Function for Locomotion Control Tasks","date":"2023-03-05","arxiv_id":"2303.02640","repositories_listed":1,"syntology":null},{"url":"/paper/corl-environment-creation-and-management","slug":"corl-environment-creation-and-management","title":"CoRL: Environment Creation and Management Focused on System Integration","date":"2023-03-03","arxiv_id":"2303.02182","repositories_listed":1,"syntology":null},{"url":"/paper/ghq-grouped-hybrid-q-learning-for","slug":"ghq-grouped-hybrid-q-learning-for","title":"GHQ: Grouped Hybrid Q Learning for Heterogeneous Cooperative Multi-agent Reinforcement Learning","date":"2023-03-02","arxiv_id":"2303.01070","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-guided-multi-objective","slug":"reinforcement-learning-guided-multi-objective","title":"Reinforcement Learning Guided Multi-Objective Exam Paper Generation","date":"2023-03-02","arxiv_id":"2303.01042","repositories_listed":1,"syntology":null},{"url":"/paper/ls-iq-implicit-reward-regularization-for","slug":"ls-iq-implicit-reward-regularization-for","title":"LS-IQ: Implicit Reward Regularization for Inverse Reinforcement Learning","date":"2023-03-01","arxiv_id":"2303.00599","repositories_listed":1,"syntology":null},{"url":"/paper/human-inspired-framework-to-accelerate","slug":"human-inspired-framework-to-accelerate","title":"Human-Inspired Framework to Accelerate Reinforcement Learning","date":"2023-02-28","arxiv_id":"2303.08115","repositories_listed":1,"syntology":null},{"url":"/paper/evotorch-scalable-evolutionary-computation-in","slug":"evotorch-scalable-evolutionary-computation-in","title":"EvoTorch: Scalable Evolutionary Computation in Python","date":"2023-02-24","arxiv_id":"2302.12600","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/evotorch-scalable-evolutionary-computation-in#ran","syntology_url":"https://syntology.ai/paper/2302.12600","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12600"}},"official":{"repos":["nnaisense/evotorch"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/model-based-uncertainty-in-value-functions","slug":"model-based-uncertainty-in-value-functions","title":"Model-Based Uncertainty in Value Functions","date":"2023-02-24","arxiv_id":"2302.12526","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/model-based-uncertainty-in-value-functions#ran","syntology_url":"https://syntology.ai/paper/2302.12526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12526"}},"official":{"repos":["boschresearch/ube-mbrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/provably-efficient-neural-offline","slug":"provably-efficient-neural-offline","title":"VIPeR: Provably Efficient Algorithm for Offline RL with Neural Function Approximation","date":"2023-02-24","arxiv_id":"2302.12780","repositories_listed":1,"syntology":null},{"url":"/paper/energy-harvesting-reconfigurable-intelligent","slug":"energy-harvesting-reconfigurable-intelligent","title":"Energy Harvesting Reconfigurable Intelligent Surface for UAV Based on Robust Deep Reinforcement Learning","date":"2023-02-23","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/trustworthy-reinforcement-learning-for","slug":"trustworthy-reinforcement-learning-for","title":"Constrained Reinforcement Learning using Distributional Representation for Trustworthy Quadrotor UAV Tracking Control","date":"2023-02-22","arxiv_id":"2302.11694","repositories_listed":1,"syntology":null},{"url":"/paper/assessment-of-reinforcement-learning-for","slug":"assessment-of-reinforcement-learning-for","title":"Assessment of Reinforcement Learning for Macro Placement","date":"2023-02-21","arxiv_id":"2302.11014","repositories_listed":1,"syntology":null},{"url":"/paper/mac-po-multi-agent-experience-replay-via","slug":"mac-po-multi-agent-experience-replay-via","title":"MAC-PO: Multi-Agent Experience Replay via Collective Priority Optimization","date":"2023-02-21","arxiv_id":"2302.10418","repositories_listed":1,"syntology":null},{"url":"/paper/minimax-bayes-reinforcement-learning","slug":"minimax-bayes-reinforcement-learning","title":"Minimax-Bayes Reinforcement Learning","date":"2023-02-21","arxiv_id":"2302.10831","repositories_listed":1,"syntology":null},{"url":"/paper/potential-based-reward-shaping-for-learning","slug":"potential-based-reward-shaping-for-learning","title":"Learning to Play Text-based Adventure Games with Maximum Entropy Reinforcement Learning","date":"2023-02-21","arxiv_id":"2302.10720","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-cost","slug":"deep-reinforcement-learning-for-cost","title":"Deep Reinforcement Learning for Cost-Effective Medical Diagnosis","date":"2023-02-20","arxiv_id":"2302.10261","repositories_listed":1,"syntology":null},{"url":"/paper/multiagent-inverse-reinforcement-learning-via","slug":"multiagent-inverse-reinforcement-learning-via","title":"Multiagent Inverse Reinforcement Learning via Theory of Mind Reasoning","date":"2023-02-20","arxiv_id":"2302.10238","repositories_listed":1,"syntology":null},{"url":"/paper/take-me-home-reversing-distribution-shifts","slug":"take-me-home-reversing-distribution-shifts","title":"DC4L: Distribution Shift Recovery via Data-Driven Control for Deep Learning Models","date":"2023-02-20","arxiv_id":"2302.10341","repositories_listed":1,"syntology":null},{"url":"/paper/generalization-in-visual-reinforcement","slug":"generalization-in-visual-reinforcement","title":"Generalization in Visual Reinforcement Learning with the Reward Sequence Distribution","date":"2023-02-19","arxiv_id":"2302.09601","repositories_listed":1,"syntology":null},{"url":"/paper/post-episodic-reinforcement-learning","slug":"post-episodic-reinforcement-learning","title":"Post Reinforcement Learning Inference","date":"2023-02-17","arxiv_id":"2302.08854","repositories_listed":1,"syntology":null},{"url":"/paper/swapped-goal-conditioned-offline","slug":"swapped-goal-conditioned-offline","title":"Swapped goal-conditioned offline reinforcement learning","date":"2023-02-17","arxiv_id":"2302.08865","repositories_listed":1,"syntology":null},{"url":"/paper/imitation-from-arbitrary-experience-a-dual","slug":"imitation-from-arbitrary-experience-a-dual","title":"Dual RL: Unification and New Methods for Reinforcement and Imitation Learning","date":"2023-02-16","arxiv_id":"2302.08560","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imitation-from-arbitrary-experience-a-dual#ran","syntology_url":"https://syntology.ai/paper/2302.08560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.08560"}},"official":{"repos":["hari-sikchi/DVL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tuning-computer-vision-models-with-task","slug":"tuning-computer-vision-models-with-task","title":"Tuning computer vision models with task rewards","date":"2023-02-16","arxiv_id":"2302.08242","repositories_listed":1,"syntology":null},{"url":"/paper/conservative-state-value-estimation-for-1","slug":"conservative-state-value-estimation-for-1","title":"Conservative State Value Estimation for Offline Reinforcement Learning","date":"2023-02-14","arxiv_id":"2302.06884","repositories_listed":1,"syntology":null},{"url":"/paper/constrained-decision-transformer-for-offline","slug":"constrained-decision-transformer-for-offline","title":"Constrained Decision Transformer for Offline Safe Reinforcement Learning","date":"2023-02-14","arxiv_id":"2302.07351","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/constrained-decision-transformer-for-offline#ran","syntology_url":"https://syntology.ai/paper/2302.07351","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.07351"}},"official":{"repos":["liuzuxin/osrl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-deep-reinforcement-learning-through-2","slug":"robust-deep-reinforcement-learning-through-2","title":"Regret-Based Defense in Adversarial Reinforcement Learning","date":"2023-02-14","arxiv_id":"2302.06912","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robust-deep-reinforcement-learning-through-2#ran","syntology_url":"https://syntology.ai/paper/2302.06912","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.06912"}},"official":{"repos":["romanbelaire/robust-ccer"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/semiconductor-fab-scheduling-with-self","slug":"semiconductor-fab-scheduling-with-self","title":"Semiconductor Fab Scheduling with Self-Supervised and Reinforcement Learning","date":"2023-02-14","arxiv_id":"2302.07162","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-noise-filtering-with-dynamic-sparse","slug":"automatic-noise-filtering-with-dynamic-sparse","title":"Automatic Noise Filtering with Dynamic Sparse Training in Deep Reinforcement Learning","date":"2023-02-13","arxiv_id":"2302.06548","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/automatic-noise-filtering-with-dynamic-sparse#ran","syntology_url":"https://syntology.ai/paper/2302.06548","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.06548"}},"official":{"repos":["bramgrooten/automatic-noise-filtering"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/guiding-pretraining-in-reinforcement-learning","slug":"guiding-pretraining-in-reinforcement-learning","title":"Guiding Pretraining in Reinforcement Learning with Large Language Models","date":"2023-02-13","arxiv_id":"2302.06692","repositories_listed":1,"syntology":null},{"url":"/paper/a-large-parametrized-space-of-meta","slug":"a-large-parametrized-space-of-meta","title":"Procedural generation of meta-reinforcement learning tasks","date":"2023-02-11","arxiv_id":"2302.05583","repositories_listed":1,"syntology":null},{"url":"/paper/cross-domain-random-pre-training-with","slug":"cross-domain-random-pre-training-with","title":"Cross-domain Random Pre-training with Prototypes for Reinforcement Learning","date":"2023-02-11","arxiv_id":"2302.05614","repositories_listed":1,"syntology":null},{"url":"/paper/a-swat-based-reinforcement-learning-framework","slug":"a-swat-based-reinforcement-learning-framework","title":"A SWAT-based Reinforcement Learning Framework for Crop Management","date":"2023-02-10","arxiv_id":"2302.04988","repositories_listed":1,"syntology":null},{"url":"/paper/on-penalty-based-bilevel-gradient-descent","slug":"on-penalty-based-bilevel-gradient-descent","title":"On Penalty-based Bilevel Gradient Descent Method","date":"2023-02-10","arxiv_id":"2302.05185","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":3,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 3 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-penalty-based-bilevel-gradient-descent#ran","syntology_url":"https://syntology.ai/paper/2302.05185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.05185"}},"official":{"repos":["hanshen95/penalized-bilevel-gradient-descent"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-from-multiple-sensors","slug":"reinforcement-learning-from-multiple-sensors","title":"Combining Reconstruction and Contrastive Methods for Multimodal Representations in RL","date":"2023-02-10","arxiv_id":"2302.05342","repositories_listed":1,"syntology":null},{"url":"/paper/robust-knowledge-transfer-in-tiered-1","slug":"robust-knowledge-transfer-in-tiered-1","title":"Robust Knowledge Transfer in Tiered Reinforcement Learning","date":"2023-02-10","arxiv_id":"2302.05534","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-knowledge-transfer-in-tiered-1#ran","syntology_url":"https://syntology.ai/paper/2302.05534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.05534"}},"official":{"repos":["jiaweihhuang/robust-tiered-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-wisdom-of-hindsight-makes-language-models","slug":"the-wisdom-of-hindsight-makes-language-models","title":"The Wisdom of Hindsight Makes Language Models Better Instruction Followers","date":"2023-02-10","arxiv_id":"2302.05206","repositories_listed":1,"syntology":null},{"url":"/paper/learning-complex-teamwork-tasks-using-a-sub","slug":"learning-complex-teamwork-tasks-using-a-sub","title":"Learning Complex Teamwork Tasks Using a Given Sub-task Decomposition","date":"2023-02-09","arxiv_id":"2302.04944","repositories_listed":1,"syntology":null},{"url":"/paper/raynet-a-simulation-platform-for-developing","slug":"raynet-a-simulation-platform-for-developing","title":"RayNet: A Simulation Platform for Developing Reinforcement Learning-Driven Network Protocols","date":"2023-02-09","arxiv_id":"2302.04519","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-goal-based-exploration-via-pruning","slug":"scaling-goal-based-exploration-via-pruning","title":"Scaling Goal-based Exploration via Pruning Proto-goals","date":"2023-02-09","arxiv_id":"2302.04693","repositories_listed":1,"syntology":null},{"url":"/paper/learning-graph-enhanced-commander-executor","slug":"learning-graph-enhanced-commander-executor","title":"Learning Graph-Enhanced Commander-Executor for Multi-Agent Navigation","date":"2023-02-08","arxiv_id":"2302.04094","repositories_listed":1,"syntology":null},{"url":"/paper/non-zero-sum-game-control-for-multi-vehicle","slug":"non-zero-sum-game-control-for-multi-vehicle","title":"Non-zero-sum Game Control for Multi-vehicle Driving via Reinforcement Learning","date":"2023-02-08","arxiv_id":"2302.03958","repositories_listed":1,"syntology":null},{"url":"/paper/attacking-cooperative-multi-agent","slug":"attacking-cooperative-multi-agent","title":"Attacking Cooperative Multi-Agent Reinforcement Learning by Adversarial Minority Influence","date":"2023-02-07","arxiv_id":"2302.03322","repositories_listed":1,"syntology":null},{"url":"/paper/multi-task-recommendations-with-reinforcement","slug":"multi-task-recommendations-with-reinforcement","title":"Multi-Task Recommendations with Reinforcement Learning","date":"2023-02-07","arxiv_id":"2302.03328","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-task-recommendations-with-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2302.03328","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.03328"}},"official":{"repos":["applied-machine-learning-lab/rmtl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/intrinsic-rewards-from-self-organizing","slug":"intrinsic-rewards-from-self-organizing","title":"Intrinsic Rewards from Self-Organizing Feature Maps for Exploration in Reinforcement Learning","date":"2023-02-06","arxiv_id":"2302.04125","repositories_listed":1,"syntology":null},{"url":"/paper/model-free-quantum-gate-design-and","slug":"model-free-quantum-gate-design-and","title":"Model-free Quantum Gate Design and Calibration using Deep Reinforcement Learning","date":"2023-02-05","arxiv_id":"2302.02371","repositories_listed":1,"syntology":null}],"record_sha256":"b155d7a12990692bbb2cbe57647869f80e4e0a3f28e3aa5bd3564ac270c33117","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}