{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/openai-gym/papers/ran/1","list_of":"/task/openai-gym","task":"OpenAI Gym","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,43],"of":43,"counts":{"archive_papers_tagged":382,"with_a_code_link":179,"where_syntology_ran_a_sample":43,"not_listed_spam_title":0,"listed":382,"listed_where_code_ran":43,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":37,"every_run_a_failure_of_syntologys_instrument":6,"listed_with_a_run_with_no_instrument_failure":37,"listed_every_run_a_failure_of_syntologys_instrument":6,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/openai-gym/papers/ran/1","prev":null,"next":null,"papers":[{"url":"/paper/2505-10861","slug":"2505-10861","title":"Improving the Data-efficiency of Reinforcement Learning by Warm-starting with LLM","date":"2025-05-16","arxiv_id":"2505.10861","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2505-10861#ran","syntology_url":"https://syntology.ai/paper/2505.10861","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.10861"}},"official":{"repos":["duongnhatthang/llamagym"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stealing-that-free-lunch-exposing-the-limits","slug":"stealing-that-free-lunch-exposing-the-limits","title":"Stealing That Free Lunch: Exposing the Limits of Dyna-Style Reinforcement Learning","date":"2024-12-18","arxiv_id":"2412.14312","repositories_listed":0,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/stealing-that-free-lunch-exposing-the-limits#ran","syntology_url":"https://syntology.ai/paper/2412.14312","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.14312"}},"official":null}},{"url":"/paper/ompo-a-unified-framework-for-rl-under-policy","slug":"ompo-a-unified-framework-for-rl-under-policy","title":"OMPO: A Unified Framework for RL under Policy and Dynamics Shifts","date":"2024-05-29","arxiv_id":"2405.19080","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ompo-a-unified-framework-for-rl-under-policy#ran","syntology_url":"https://syntology.ai/paper/2405.19080","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19080"}},"official":{"repos":["roythuly/ompo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/llf-bench-benchmark-for-interactive-learning","slug":"llf-bench-benchmark-for-interactive-learning","title":"LLF-Bench: Benchmark for Interactive Learning from Language Feedback","date":"2023-12-11","arxiv_id":"2312.06853","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llf-bench-benchmark-for-interactive-learning#ran","syntology_url":"https://syntology.ai/paper/2312.06853","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06853"}},"official":null}},{"url":"/paper/offline-retraining-for-online-rl-decoupled","slug":"offline-retraining-for-online-rl-decoupled","title":"Offline Retraining for Online RL: Decoupled Policy Learning to Mitigate Exploration Bias","date":"2023-10-12","arxiv_id":"2310.08558","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/offline-retraining-for-online-rl-decoupled#ran","syntology_url":"https://syntology.ai/paper/2310.08558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08558"}},"official":{"repos":["MaxSobolMark/OOO"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/risk-aware-reward-shaping-of-reinforcement","slug":"risk-aware-reward-shaping-of-reinforcement","title":"Risk-Aware Reward Shaping of Reinforcement Learning Agents for Autonomous Driving","date":"2023-06-05","arxiv_id":"2306.03220","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/risk-aware-reward-shaping-of-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2306.03220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03220"}},"official":{"repos":["zhang-zengjie/code_2023_iecon_shaping_wu"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/for-sale-state-action-representation-learning-1","slug":"for-sale-state-action-representation-learning-1","title":"For SALE: State-Action Representation Learning for Deep Reinforcement Learning","date":"2023-06-04","arxiv_id":"2306.02451","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/for-sale-state-action-representation-learning-1#ran","syntology_url":"https://syntology.ai/paper/2306.02451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.02451"}},"official":{"repos":["sfujim/td7"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/average-constrained-policy-optimization","slug":"average-constrained-policy-optimization","title":"ACPO: A Policy Optimization Algorithm for Average MDPs with Constraints","date":"2023-02-02","arxiv_id":"2302.00808","repositories_listed":0,"syntology":{"n":12,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/average-constrained-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2302.00808","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00808"}},"official":null}},{"url":"/paper/evox-a-distributed-gpu-accelerated-library","slug":"evox-a-distributed-gpu-accelerated-library","title":"EvoX: A Distributed GPU-accelerated Framework for Scalable Evolutionary Computation","date":"2023-01-29","arxiv_id":"2301.12457","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evox-a-distributed-gpu-accelerated-library#ran","syntology_url":"https://syntology.ai/paper/2301.12457","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12457"}},"official":{"repos":["emi-group/evox"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-policy-optimization-in-deep","slug":"robust-policy-optimization-in-deep","title":"Robust Policy Optimization in Deep Reinforcement Learning","date":"2022-12-14","arxiv_id":"2212.07536","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-policy-optimization-in-deep#ran","syntology_url":"https://syntology.ai/paper/2212.07536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.07536"}},"official":{"repos":["vwxyzjn/cleanrl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/pyrddlgym-from-rddl-to-gym-environments","slug":"pyrddlgym-from-rddl-to-gym-environments","title":"pyRDDLGym: From RDDL to Gym Environments","date":"2022-11-11","arxiv_id":"2211.05939","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pyrddlgym-from-rddl-to-gym-environments#ran","syntology_url":"https://syntology.ai/paper/2211.05939","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.05939"}},"official":{"repos":["ataitler/pyrddlgym","pyrddlgym-project/pyrddlgym"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/quantum-deep-reinforcement-learning-for-robot","slug":"quantum-deep-reinforcement-learning-for-robot","title":"Quantum Deep Reinforcement Learning for Robot Navigation Tasks","date":"2022-02-24","arxiv_id":"2202.12180","repositories_listed":2,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/quantum-deep-reinforcement-learning-for-robot#ran","syntology_url":"https://syntology.ai/paper/2202.12180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.12180"}},"official":{"repos":["dfki-ric-quantum/qdrl-turtlebot-env","dfki-ric-quantum/qdrl-turtlebot-eval"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptively-calibrated-critic-estimates-for","slug":"adaptively-calibrated-critic-estimates-for","title":"Adaptively Calibrated Critic Estimates for Deep Reinforcement Learning","date":"2021-11-24","arxiv_id":"2111.12673","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptively-calibrated-critic-estimates-for#ran","syntology_url":"https://syntology.ai/paper/2111.12673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.12673"}},"official":{"repos":["nicolinho/acc"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stackelberg-actor-critic-game-theoretic","slug":"stackelberg-actor-critic-game-theoretic","title":"Stackelberg Actor-Critic: Game-Theoretic Reinforcement Learning Algorithms","date":"2021-09-25","arxiv_id":"2109.12286","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stackelberg-actor-critic-game-theoretic#ran","syntology_url":"https://syntology.ai/paper/2109.12286","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.12286"}},"official":{"repos":["leozhengzly/stackelberg-actor-critic-algos"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/compilergym-robust-performant-compiler","slug":"compilergym-robust-performant-compiler","title":"CompilerGym: Robust, Performant Compiler Optimization Environments for AI Research","date":"2021-09-17","arxiv_id":"2109.08267","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/compilergym-robust-performant-compiler#ran","syntology_url":"https://syntology.ai/paper/2109.08267","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.08267"}},"official":{"repos":["facebookresearch/CompilerGym"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/powergym-a-reinforcement-learning-environment","slug":"powergym-a-reinforcement-learning-environment","title":"PowerGym: A Reinforcement Learning Environment for Volt-Var Control in Power Distribution Systems","date":"2021-09-08","arxiv_id":"2109.03970","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/powergym-a-reinforcement-learning-environment#ran","syntology_url":"https://syntology.ai/paper/2109.03970","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.03970"}},"official":{"repos":["siemens/powergym"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/catastrophic-interference-in-reinforcement","slug":"catastrophic-interference-in-reinforcement","title":"Catastrophic Interference in Reinforcement Learning: A Solution Based on Context Division and Knowledge Distillation","date":"2021-09-01","arxiv_id":"2109.00525","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/catastrophic-interference-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2109.00525","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.00525"}},"official":{"repos":["sweety-dm/interference-aware-deep-q-learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-goal-reinforcement-learning","slug":"multi-goal-reinforcement-learning","title":"Multi-Goal Reinforcement Learning environments for simulated Franka Emika Panda robot","date":"2021-06-25","arxiv_id":"2106.13687","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-goal-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2106.13687","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.13687"}},"official":{"repos":["qgallouedec/panda-gym"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/brax-a-differentiable-physics-engine-for","slug":"brax-a-differentiable-physics-engine-for","title":"Brax -- A Differentiable Physics Engine for Large Scale Rigid Body Simulation","date":"2021-06-24","arxiv_id":"2106.13281","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/brax-a-differentiable-physics-engine-for#ran","syntology_url":"https://syntology.ai/paper/2106.13281","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.13281"}},"official":{"repos":["google/brax"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-sparse-training-for-deep","slug":"dynamic-sparse-training-for-deep","title":"Dynamic Sparse Training for Deep Reinforcement Learning","date":"2021-06-08","arxiv_id":"2106.04217","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-sparse-training-for-deep#ran","syntology_url":"https://syntology.ai/paper/2106.04217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04217"}},"official":{"repos":["GhadaSokar/Dynamic-Sparse-Training-for-Deep-Reinforcement-Learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/decision-transformer-reinforcement-learning","slug":"decision-transformer-reinforcement-learning","title":"Decision Transformer: Reinforcement Learning via Sequence Modeling","date":"2021-06-02","arxiv_id":"2106.01345","repositories_listed":20,"syntology":{"n":26,"n_ran":17,"n_constructed":10,"n_ran_checked":13,"n_instrument":4,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":8,"phrase":"17 ran (of which 10 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 4 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/decision-transformer-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2106.01345","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.01345"}},"official":{"repos":["kzl/decision-transformer"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/tonic-a-deep-reinforcement-learning-library","slug":"tonic-a-deep-reinforcement-learning-library","title":"Tonic: A Deep Reinforcement Learning Library for Fast Prototyping and Benchmarking","date":"2020-11-15","arxiv_id":"2011.07537","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tonic-a-deep-reinforcement-learning-library#ran","syntology_url":"https://syntology.ai/paper/2011.07537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.07537"}},"official":{"repos":["fabiopardo/tonic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ecole-a-gym-like-library-for-machine-learning","slug":"ecole-a-gym-like-library-for-machine-learning","title":"Ecole: A Gym-like Library for Machine Learning in Combinatorial Optimization Solvers","date":"2020-11-11","arxiv_id":"2011.06069","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ecole-a-gym-like-library-for-machine-learning#ran","syntology_url":"https://syntology.ai/paper/2011.06069","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.06069"}},"official":{"repos":["ds4dm/ecole"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/deep-reinforcement-learning-with-population","slug":"deep-reinforcement-learning-with-population","title":"Deep Reinforcement Learning with Population-Coded Spiking Neural Network for Continuous Control","date":"2020-10-19","arxiv_id":"2010.09635","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-with-population#ran","syntology_url":"https://syntology.ai/paper/2010.09635","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09635"}},"official":{"repos":["combra-lab/pop-spiking-deep-rl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/implicit-distributional-reinforcement","slug":"implicit-distributional-reinforcement","title":"Implicit Distributional Reinforcement Learning","date":"2020-07-13","arxiv_id":"2007.06159","repositories_listed":3,"syntology":{"n":6,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/implicit-distributional-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2007.06159","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.06159"}},"official":{"repos":["zhougroup/IDAC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/otoworld-towards-learning-to-separate-by","slug":"otoworld-towards-learning-to-separate-by","title":"OtoWorld: Towards Learning to Separate by Learning to Move","date":"2020-07-12","arxiv_id":"2007.06123","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/otoworld-towards-learning-to-separate-by#ran","syntology_url":"https://syntology.ai/paper/2007.06123","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.06123"}},"official":{"repos":["pseeth/otoworld"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/experience-replay-with-likelihood-free","slug":"experience-replay-with-likelihood-free","title":"Experience Replay with Likelihood-free Importance Weights","date":"2020-06-23","arxiv_id":"2006.13169","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/experience-replay-with-likelihood-free#ran","syntology_url":"https://syntology.ai/paper/2006.13169","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.13169"}},"official":null}},{"url":"/paper/reinforcement-learning-with-augmented-data","slug":"reinforcement-learning-with-augmented-data","title":"Reinforcement Learning with Augmented Data","date":"2020-04-30","arxiv_id":"2004.14990","repositories_listed":2,"syntology":{"n":24,"n_ran":21,"n_constructed":8,"n_ran_checked":9,"n_instrument":12,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":21,"phrase":"21 ran (of which 8 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 12 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reinforcement-learning-with-augmented-data#ran","syntology_url":"https://syntology.ai/paper/2004.14990","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.14990"}},"official":null}},{"url":"/paper/state-only-imitation-with-transition-dynamics-1","slug":"state-only-imitation-with-transition-dynamics-1","title":"State-only Imitation with Transition Dynamics Mismatch","date":"2020-02-27","arxiv_id":"2002.11879","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/state-only-imitation-with-transition-dynamics-1#ran","syntology_url":"https://syntology.ai/paper/2002.11879","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.11879"}},"official":{"repos":["tgangwani/RL-Indirect-imitation"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pddlgym-gym-environments-from-pddl-problems","slug":"pddlgym-gym-environments-from-pddl-problems","title":"PDDLGym: Gym Environments from PDDL Problems","date":"2020-02-15","arxiv_id":"2002.06432","repositories_listed":2,"syntology":{"n":11,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/pddlgym-gym-environments-from-pddl-problems#ran","syntology_url":"https://syntology.ai/paper/2002.06432","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.06432"}},"official":{"repos":["tomsilver/pddlgym"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/discrete-action-on-policy-learning-with","slug":"discrete-action-on-policy-learning-with","title":"Discrete Action On-Policy Learning with Action-Value Critic","date":"2020-02-10","arxiv_id":"2002.03534","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/discrete-action-on-policy-learning-with#ran","syntology_url":"https://syntology.ai/paper/2002.03534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.03534"}},"official":{"repos":["yuguangyue/CARSM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/slm-lab-a-comprehensive-benchmark-and-modular-1","slug":"slm-lab-a-comprehensive-benchmark-and-modular-1","title":"SLM Lab: A Comprehensive Benchmark and Modular Software Framework for Reproducible Deep Reinforcement Learning","date":"2019-12-28","arxiv_id":"1912.12482","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/slm-lab-a-comprehensive-benchmark-and-modular-1#ran","syntology_url":"https://syntology.ai/paper/1912.12482","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.12482"}},"official":{"repos":["kengz/SLM-Lab"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/playing-games-in-the-dark-an-approach-for","slug":"playing-games-in-the-dark-an-approach-for","title":"Playing Games in the Dark: An approach for cross-modality transfer in reinforcement learning","date":"2019-11-28","arxiv_id":"1911.12851","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":3,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/playing-games-in-the-dark-an-approach-for#ran","syntology_url":"https://syntology.ai/paper/1911.12851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.12851"}},"official":null}},{"url":"/paper/torchbeast-a-pytorch-platform-for-distributed","slug":"torchbeast-a-pytorch-platform-for-distributed","title":"TorchBeast: A PyTorch Platform for Distributed RL","date":"2019-10-08","arxiv_id":"1910.03552","repositories_listed":3,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/torchbeast-a-pytorch-platform-for-distributed#ran","syntology_url":"https://syntology.ai/paper/1910.03552","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.03552"}},"official":{"repos":["heiner/scalable_agent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/v-mpo-on-policy-maximum-a-posteriori-policy","slug":"v-mpo-on-policy-maximum-a-posteriori-policy","title":"V-MPO: On-Policy Maximum a Posteriori Policy Optimization for Discrete and Continuous Control","date":"2019-09-26","arxiv_id":"1909.12238","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/v-mpo-on-policy-maximum-a-posteriori-policy#ran","syntology_url":"https://syntology.ai/paper/1909.12238","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.12238"}},"official":null}},{"url":"/paper/mdp-playground-meta-features-in-reinforcement","slug":"mdp-playground-meta-features-in-reinforcement","title":"MDP Playground: An Analysis and Debug Testbed for Reinforcement Learning","date":"2019-09-17","arxiv_id":"1909.07750","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mdp-playground-meta-features-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1909.07750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.07750"}},"official":{"repos":["automl/mdp-playground"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/proximal-distilled-evolutionary-reinforcement","slug":"proximal-distilled-evolutionary-reinforcement","title":"Proximal Distilled Evolutionary Reinforcement Learning","date":"2019-06-24","arxiv_id":"1906.09807","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/proximal-distilled-evolutionary-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1906.09807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.09807"}},"official":{"repos":["crisbodnar/pderl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-transfer-between-atari-games-using","slug":"visual-transfer-between-atari-games-using","title":"Visual Transfer between Atari Games using Competitive Reinforcement Learning","date":"2018-09-02","arxiv_id":"1809.00397","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/visual-transfer-between-atari-games-using#ran","syntology_url":"https://syntology.ai/paper/1809.00397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.00397"}},"official":{"repos":["sowmya-mp/rl_a3c_pytorch"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-general-video","slug":"deep-reinforcement-learning-for-general-video","title":"Deep Reinforcement Learning for General Video Game AI","date":"2018-06-06","arxiv_id":"1806.02448","repositories_listed":2,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-for-general-video#ran","syntology_url":"https://syntology.ai/paper/1806.02448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.02448"}},"official":{"repos":["rubenrtorrado/GVGAI_GYM"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/addressing-function-approximation-error-in","slug":"addressing-function-approximation-error-in","title":"Addressing Function Approximation Error in Actor-Critic Methods","date":"2018-02-26","arxiv_id":"1802.09477","repositories_listed":67,"syntology":{"n":36,"n_ran":26,"n_constructed":0,"n_ran_checked":25,"n_instrument":1,"n_unverified":10,"n_honours":1,"n_violates":1,"n_no_contract":23,"n_pointer_only":21,"phrase":"26 ran (of which 0 constructed an object rather than computing a result; 25 with no instrument failure: 1 honoured, 1 violated, 23 with no contract checked; 1 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/addressing-function-approximation-error-in#ran","syntology_url":"https://syntology.ai/paper/1802.09477","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.09477"}},"official":{"repos":["sfujim/TD3"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/soft-actor-critic-off-policy-maximum-entropy","slug":"soft-actor-critic-off-policy-maximum-entropy","title":"Soft Actor-Critic: Off-Policy Maximum Entropy Deep Reinforcement Learning with a Stochastic Actor","date":"2018-01-04","arxiv_id":"1801.01290","repositories_listed":86,"syntology":{"n":148,"n_ran":91,"n_constructed":61,"n_ran_checked":73,"n_instrument":18,"n_unverified":57,"n_honours":3,"n_violates":1,"n_no_contract":69,"n_pointer_only":66,"phrase":"91 ran (of which 61 constructed an object rather than computing a result; 73 with no instrument failure: 3 honoured, 1 violated, 69 with no contract checked; 18 where Syntology's instrument failed) · 57 unverified","sample_list":"/paper/soft-actor-critic-off-policy-maximum-entropy#ran","syntology_url":"https://syntology.ai/paper/1801.01290","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1801.01290"}},"official":{"repos":["haarnoja/sac"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/proximal-policy-optimization-algorithms","slug":"proximal-policy-optimization-algorithms","title":"Proximal Policy Optimization Algorithms","date":"2017-07-20","arxiv_id":"1707.06347","repositories_listed":188,"syntology":{"n":176,"n_ran":99,"n_constructed":53,"n_ran_checked":71,"n_instrument":28,"n_unverified":77,"n_honours":7,"n_violates":2,"n_no_contract":62,"n_pointer_only":94,"phrase":"99 ran (of which 53 constructed an object rather than computing a result; 71 with no instrument failure: 7 honoured, 2 violated, 62 with no contract checked; 28 where Syntology's instrument failed) · 77 unverified","sample_list":"/paper/proximal-policy-optimization-algorithms#ran","syntology_url":"https://syntology.ai/paper/1707.06347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1707.06347"}},"official":null}},{"url":"/paper/continuous-control-with-deep-reinforcement","slug":"continuous-control-with-deep-reinforcement","title":"Continuous control with deep reinforcement learning","date":"2015-09-09","arxiv_id":"1509.02971","repositories_listed":161,"syntology":{"n":306,"n_ran":163,"n_constructed":126,"n_ran_checked":152,"n_instrument":11,"n_unverified":143,"n_honours":3,"n_violates":0,"n_no_contract":149,"n_pointer_only":163,"phrase":"163 ran (of which 126 constructed an object rather than computing a result; 152 with no instrument failure: 3 honoured, 0 violated, 149 with no contract checked; 11 where Syntology's instrument failed) · 143 unverified","sample_list":"/paper/continuous-control-with-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1509.02971","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1509.02971"}},"official":null}}],"record_sha256":"4ddce629f16bc92da4f5e41ce895f204b4907cbf5f965d64f82718b5968b0ba8","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}