{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/6","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":152,"rows_per_page":100,"rows":[501,600],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/5","next":"/task/reinforcement-learning-1/papers/7","papers":[{"url":"/paper/heterogeneous-multi-robot-reinforcement","slug":"heterogeneous-multi-robot-reinforcement","title":"Heterogeneous Multi-Robot Reinforcement Learning","date":"2023-01-17","arxiv_id":"2301.07137","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/heterogeneous-multi-robot-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2301.07137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.07137"}},"official":{"repos":["proroklab/hetgppo","proroklab/vectorizedmultiagentsimulator"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/asynchronous-multi-agent-reinforcement","slug":"asynchronous-multi-agent-reinforcement","title":"Asynchronous Multi-Agent Reinforcement Learning for Efficient Real-Time Multi-Robot Cooperative Exploration","date":"2023-01-09","arxiv_id":"2301.03398","repositories_listed":2,"syntology":null},{"url":"/paper/automated-gadget-discovery-in-science","slug":"automated-gadget-discovery-in-science","title":"Automated Gadget Discovery in Science","date":"2022-12-24","arxiv_id":"2212.12743","repositories_listed":2,"syntology":null},{"url":"/paper/targeted-adversarial-attacks-on-deep","slug":"targeted-adversarial-attacks-on-deep","title":"Targeted Adversarial Attacks on Deep Reinforcement Learning Policies via Model Checking","date":"2022-12-10","arxiv_id":"2212.05337","repositories_listed":2,"syntology":null},{"url":"/paper/compiler-optimization-for-quantum-computing","slug":"compiler-optimization-for-quantum-computing","title":"Compiler Optimization for Quantum Computing Using Reinforcement Learning","date":"2022-12-08","arxiv_id":"2212.04508","repositories_listed":2,"syntology":null},{"url":"/paper/mo-gym-a-library-of-multi-objective","slug":"mo-gym-a-library-of-multi-objective","title":"MO-Gym: A Library of Multi-Objective Reinforcement Learning Environments","date":"2022-11-30","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/the-surprising-effectiveness-of-latent-world","slug":"the-surprising-effectiveness-of-latent-world","title":"The Effectiveness of World Models for Continual Reinforcement Learning","date":"2022-11-29","arxiv_id":"2211.15944","repositories_listed":2,"syntology":null},{"url":"/paper/pac-man-pete-an-extensible-framework-for","slug":"pac-man-pete-an-extensible-framework-for","title":"Pac-Man Pete: An extensible framework for building AI in VEX Robotics","date":"2022-11-25","arxiv_id":"2211.14385","repositories_listed":2,"syntology":null},{"url":"/paper/imitation-clean-imitation-learning","slug":"imitation-clean-imitation-learning","title":"imitation: Clean Imitation Learning Implementations","date":"2022-11-22","arxiv_id":"2211.11972","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imitation-clean-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/2211.11972","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.11972"}},"official":{"repos":["HumanCompatibleAI/imitation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/let-offline-rl-flow-training-conservative","slug":"let-offline-rl-flow-training-conservative","title":"Let Offline RL Flow: Training Conservative Agents in the Latent Space of Normalizing Flows","date":"2022-11-20","arxiv_id":"2211.11096","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/let-offline-rl-flow-training-conservative#ran","syntology_url":"https://syntology.ai/paper/2211.11096","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.11096"}},"official":{"repos":["tinkoff-ai/cnf"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/libsignal-an-open-library-for-traffic-signal","slug":"libsignal-an-open-library-for-traffic-signal","title":"LibSignal: An Open Library for Traffic Signal Control","date":"2022-11-19","arxiv_id":"2211.10649","repositories_listed":2,"syntology":null},{"url":"/paper/protox-explaining-a-reinforcement-learning","slug":"protox-explaining-a-reinforcement-learning","title":"ProtoX: Explaining a Reinforcement Learning Agent via Prototyping","date":"2022-11-06","arxiv_id":"2211.03162","repositories_listed":2,"syntology":null},{"url":"/paper/scalable-multi-agent-reinforcement-learning-2","slug":"scalable-multi-agent-reinforcement-learning-2","title":"Scalable Multi-Agent Reinforcement Learning through Intelligent Information Aggregation","date":"2022-11-03","arxiv_id":"2211.02127","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scalable-multi-agent-reinforcement-learning-2#ran","syntology_url":"https://syntology.ai/paper/2211.02127","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.02127"}},"official":{"repos":["nsidn98/informarl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/causal-counterfactuals-for-improving-the","slug":"causal-counterfactuals-for-improving-the","title":"Causal Counterfactuals for Improving the Robustness of Reinforcement Learning","date":"2022-11-02","arxiv_id":"2211.05551","repositories_listed":2,"syntology":null},{"url":"/paper/agent-controller-representations-principled","slug":"agent-controller-representations-principled","title":"Agent-Controller Representations: Principled Offline RL with Rich Exogenous Information","date":"2022-10-31","arxiv_id":"2211.00164","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":6,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 6 samples that ran constructed an object rather than computing a result","sample_list":"/paper/agent-controller-representations-principled#ran","syntology_url":"https://syntology.ai/paper/2211.00164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.00164"}},"official":{"repos":["manantomar/agent-centric-representations"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":6,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/defix-detecting-and-fixing-failure-scenarios","slug":"defix-detecting-and-fixing-failure-scenarios","title":"DeFIX: Detecting and Fixing Failure Scenarios with Reinforcement Learning in Imitation Learning Based Autonomous Driving","date":"2022-10-29","arxiv_id":"2210.16567","repositories_listed":2,"syntology":null},{"url":"/paper/self-improving-safety-performance-of","slug":"self-improving-safety-performance-of","title":"Self-Improving Safety Performance of Reinforcement Learning Based Driving with Black-Box Verification Algorithms","date":"2022-10-29","arxiv_id":"2210.16575","repositories_listed":2,"syntology":null},{"url":"/paper/adaptive-behavior-cloning-regularization-for-1","slug":"adaptive-behavior-cloning-regularization-for-1","title":"Adaptive Behavior Cloning Regularization for Stable Offline-to-Online Reinforcement Learning","date":"2022-10-25","arxiv_id":"2210.13846","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptive-behavior-cloning-regularization-for-1#ran","syntology_url":"https://syntology.ai/paper/2210.13846","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13846"}},"official":{"repos":["zhaoyi11/adaptive_bc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dextreme-transfer-of-agile-in-hand","slug":"dextreme-transfer-of-agile-in-hand","title":"DeXtreme: Transfer of Agile In-hand Manipulation from Simulation to Reality","date":"2022-10-25","arxiv_id":"2210.13702","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dextreme-transfer-of-agile-in-hand#ran","syntology_url":"https://syntology.ai/paper/2210.13702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13702"}},"official":{"repos":["Denys88/rl_games","NVIDIA-Omniverse/IsaacGymEnvs"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-lifelong-reinforcement-learning","slug":"model-based-lifelong-reinforcement-learning","title":"Model-based Lifelong Reinforcement Learning with Bayesian Exploration","date":"2022-10-20","arxiv_id":"2210.11579","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-based-lifelong-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2210.11579","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.11579"}},"official":{"repos":["minusadd/vblrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a-multilevel-reinforcement-learning-framework","slug":"a-multilevel-reinforcement-learning-framework","title":"A Multilevel Reinforcement Learning Framework for PDE-based Control","date":"2022-10-15","arxiv_id":"2210.08400","repositories_listed":2,"syntology":null},{"url":"/paper/exploration-via-elliptical-episodic-bonuses","slug":"exploration-via-elliptical-episodic-bonuses","title":"Exploration via Elliptical Episodic Bonuses","date":"2022-10-11","arxiv_id":"2210.05805","repositories_listed":2,"syntology":null},{"url":"/paper/exploration-via-planning-for-information","slug":"exploration-via-planning-for-information","title":"Exploration via Planning for Information about the Optimal Trajectory","date":"2022-10-06","arxiv_id":"2210.04642","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":6,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 6 samples that ran constructed an object rather than computing a result","sample_list":"/paper/exploration-via-planning-for-information#ran","syntology_url":"https://syntology.ai/paper/2210.04642","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.04642"}},"official":{"repos":["fusion-ml/trajectory-information-rl"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/discovering-faster-matrix-multiplication","slug":"discovering-faster-matrix-multiplication","title":"Discovering faster matrix multiplication algorithms with reinforcement learning","date":"2022-10-05","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/real-time-reinforcement-learning-for-vision","slug":"real-time-reinforcement-learning-for-vision","title":"Real-Time Reinforcement Learning for Vision-Based Robotics Utilizing Local and Remote Computers","date":"2022-10-05","arxiv_id":"2210.02317","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/real-time-reinforcement-learning-for-vision#ran","syntology_url":"https://syntology.ai/paper/2210.02317","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.02317"}},"official":{"repos":["rlai-lab/relod","rlai-lab/remote-onboard-agent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stateful-active-facilitator-coordination-and","slug":"stateful-active-facilitator-coordination-and","title":"Stateful active facilitator: Coordination and Environmental Heterogeneity in Cooperative Multi-Agent Reinforcement Learning","date":"2022-10-04","arxiv_id":"2210.03022","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stateful-active-facilitator-coordination-and#ran","syntology_url":"https://syntology.ai/paper/2210.03022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03022"}},"official":{"repos":["jaggbow/saf","veds12/hecogrid"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-intrinsically-motivated-exploration-in","slug":"deep-intrinsically-motivated-exploration-in","title":"Deep Intrinsically Motivated Exploration in Continuous Control","date":"2022-10-01","arxiv_id":"2210.00293","repositories_listed":2,"syntology":null},{"url":"/paper/on-efficient-reinforcement-learning-for-full","slug":"on-efficient-reinforcement-learning-for-full","title":"On Efficient Reinforcement Learning for Full-length Game of StarCraft II","date":"2022-09-23","arxiv_id":"2209.11553","repositories_listed":2,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/on-efficient-reinforcement-learning-for-full#ran","syntology_url":"https://syntology.ai/paper/2209.11553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.11553"}},"official":{"repos":["liuruoze/hiernet-sc2","liuruoze/mini-AlphaStar"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-gym-environments-for-limit-order","slug":"model-based-gym-environments-for-limit-order","title":"Model-based gym environments for limit order book trading","date":"2022-09-16","arxiv_id":"2209.07823","repositories_listed":2,"syntology":null},{"url":"/paper/cool-mc-a-comprehensive-tool-for","slug":"cool-mc-a-comprehensive-tool-for","title":"COOL-MC: A Comprehensive Tool for Reinforcement Learning and Model Checking","date":"2022-09-15","arxiv_id":"2209.07133","repositories_listed":2,"syntology":null},{"url":"/paper/white-box-adversarial-policies-in-deep","slug":"white-box-adversarial-policies-in-deep","title":"Red Teaming with Mind Reading: White-Box Adversarial Policies Against RL Agents","date":"2022-09-05","arxiv_id":"2209.02167","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/white-box-adversarial-policies-in-deep#ran","syntology_url":"https://syntology.ai/paper/2209.02167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.02167"}},"official":{"repos":["thestephencasper/lm_white_box_attacks","thestephencasper/white_box_rarl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/transformers-are-sample-efficient-world","slug":"transformers-are-sample-efficient-world","title":"Transformers are Sample-Efficient World Models","date":"2022-09-01","arxiv_id":"2209.00588","repositories_listed":2,"syntology":{"n":26,"n_ran":17,"n_constructed":13,"n_ran_checked":16,"n_instrument":1,"n_unverified":9,"n_honours":1,"n_violates":0,"n_no_contract":15,"n_pointer_only":26,"phrase":"17 ran (of which 13 constructed an object rather than computing a result; 16 with no instrument failure: 1 honoured, 0 violated, 15 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/transformers-are-sample-efficient-world#ran","syntology_url":"https://syntology.ai/paper/2209.00588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.00588"}},"official":{"repos":["eloialonso/iris"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":13,"n_ran_no_instrument_failure":16,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-automated-imbalanced-learning-with","slug":"towards-automated-imbalanced-learning-with","title":"Towards Automated Imbalanced Learning with Deep Hierarchical Reinforcement Learning","date":"2022-08-26","arxiv_id":"2208.12433","repositories_listed":2,"syntology":null},{"url":"/paper/solving-royal-game-of-ur-using-reinforcement","slug":"solving-royal-game-of-ur-using-reinforcement","title":"Solving Royal Game of Ur Using Reinforcement Learning","date":"2022-08-23","arxiv_id":"2208.10669","repositories_listed":2,"syntology":null},{"url":"/paper/incorporating-rivalry-in-reinforcement-1","slug":"incorporating-rivalry-in-reinforcement-1","title":"Incorporating Rivalry in Reinforcement Learning for a Competitive Game","date":"2022-08-22","arxiv_id":"2208.10327","repositories_listed":2,"syntology":null},{"url":"/paper/metric-residual-networks-for-sample-efficient","slug":"metric-residual-networks-for-sample-efficient","title":"Metric Residual Networks for Sample Efficient Goal-Conditioned Reinforcement Learning","date":"2022-08-17","arxiv_id":"2208.08133","repositories_listed":2,"syntology":null},{"url":"/paper/bsac-bayesian-strategy-network-based-soft","slug":"bsac-bayesian-strategy-network-based-soft","title":"Bayesian Soft Actor-Critic: A Directed Acyclic Strategy Graph Based Deep Reinforcement Learning","date":"2022-08-11","arxiv_id":"2208.06033","repositories_listed":2,"syntology":null},{"url":"/paper/towards-sequence-level-training-for-visual","slug":"towards-sequence-level-training-for-visual","title":"Towards Sequence-Level Training for Visual Tracking","date":"2022-08-11","arxiv_id":"2208.05810","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-sequence-level-training-for-visual#ran","syntology_url":"https://syntology.ai/paper/2208.05810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.05810"}},"official":{"repos":["byminji/SLTtrack"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/automating-dbscan-via-deep-reinforcement","slug":"automating-dbscan-via-deep-reinforcement","title":"Automating DBSCAN via Deep Reinforcement Learning","date":"2022-08-09","arxiv_id":"2208.04537","repositories_listed":2,"syntology":null},{"url":"/paper/hierarchical-kickstarting-for-skill-transfer","slug":"hierarchical-kickstarting-for-skill-transfer","title":"Hierarchical Kickstarting for Skill Transfer in Reinforcement Learning","date":"2022-07-23","arxiv_id":"2207.11584","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hierarchical-kickstarting-for-skill-transfer#ran","syntology_url":"https://syntology.ai/paper/2207.11584","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.11584"}},"official":{"repos":["ucl-dark/skillhack"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/log-barriers-for-safe-black-box-optimization","slug":"log-barriers-for-safe-black-box-optimization","title":"Log Barriers for Safe Black-box Optimization with Application to Safe Reinforcement Learning","date":"2022-07-21","arxiv_id":"2207.10415","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/log-barriers-for-safe-black-box-optimization#ran","syntology_url":"https://syntology.ai/paper/2207.10415","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.10415"}},"official":{"repos":["ilnura/lb_sgd","lasgroup/lbsgd-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/discriminator-weighted-offline-imitation-1","slug":"discriminator-weighted-offline-imitation-1","title":"Discriminator-Weighted Offline Imitation Learning from Suboptimal Demonstrations","date":"2022-07-20","arxiv_id":"2207.10050","repositories_listed":2,"syntology":{"n":13,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":13,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/discriminator-weighted-offline-imitation-1#ran","syntology_url":"https://syntology.ai/paper/2207.10050","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.10050"}},"official":{"repos":["ryanxhr/dwbc"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/bayesian-generational-population-based","slug":"bayesian-generational-population-based","title":"Bayesian Generational Population-Based Training","date":"2022-07-19","arxiv_id":"2207.09405","repositories_listed":2,"syntology":null},{"url":"/paper/dgpo-discovering-multiple-strategies-with","slug":"dgpo-discovering-multiple-strategies-with","title":"DGPO: Discovering Multiple Strategies with Diversity-Guided Policy Optimization","date":"2022-07-12","arxiv_id":"2207.05631","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dgpo-discovering-multiple-strategies-with#ran","syntology_url":"https://syntology.ai/paper/2207.05631","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05631"}},"official":{"repos":["OpenRL-Lab/DGPO"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/coderl-mastering-code-generation-through","slug":"coderl-mastering-code-generation-through","title":"CodeRL: Mastering Code Generation through Pretrained Models and Deep Reinforcement Learning","date":"2022-07-05","arxiv_id":"2207.01780","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coderl-mastering-code-generation-through#ran","syntology_url":"https://syntology.ai/paper/2207.01780","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.01780"}},"official":{"repos":["salesforce/coderl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/goal-conditioned-generators-of-deep-policies","slug":"goal-conditioned-generators-of-deep-policies","title":"Goal-Conditioned Generators of Deep Policies","date":"2022-07-04","arxiv_id":"2207.01570","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/goal-conditioned-generators-of-deep-policies#ran","syntology_url":"https://syntology.ai/paper/2207.01570","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.01570"}},"official":{"repos":["idsia/gogepo"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/on-the-learning-and-learnablity-of","slug":"on-the-learning-and-learnablity-of","title":"On the Learning and Learnability of Quasimetrics","date":"2022-06-30","arxiv_id":"2206.15478","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/on-the-learning-and-learnablity-of#ran","syntology_url":"https://syntology.ai/paper/2206.15478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.15478"}},"official":{"repos":["ssnl/poisson_quasimetric_embedding"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/video-pretraining-vpt-learning-to-act-by","slug":"video-pretraining-vpt-learning-to-act-by","title":"Video PreTraining (VPT): Learning to Act by Watching Unlabeled Online Videos","date":"2022-06-23","arxiv_id":"2206.11795","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-pretraining-vpt-learning-to-act-by#ran","syntology_url":"https://syntology.ai/paper/2206.11795","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.11795"}},"official":{"repos":["openai/Video-Pre-Training"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-constraint-inference-in-inverse","slug":"benchmarking-constraint-inference-in-inverse","title":"Benchmarking Constraint Inference in Inverse Reinforcement Learning","date":"2022-06-20","arxiv_id":"2206.09670","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-constraint-inference-in-inverse#ran","syntology_url":"https://syntology.ai/paper/2206.09670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.09670"}},"official":{"repos":["guiliang/cirl-benchmarks-public","guiliang/icrl-benchmarks-public"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/anchor-changing-regularized-natural-policy","slug":"anchor-changing-regularized-natural-policy","title":"Anchor-Changing Regularized Natural Policy Gradient for Multi-Objective Reinforcement Learning","date":"2022-06-10","arxiv_id":"2206.05357","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/anchor-changing-regularized-natural-policy#ran","syntology_url":"https://syntology.ai/paper/2206.05357","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05357"}},"official":{"repos":["tliu1997/arnpg-morl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/does-self-supervised-learning-really-improve","slug":"does-self-supervised-learning-really-improve","title":"Does Self-supervised Learning Really Improve Reinforcement Learning from Pixels?","date":"2022-06-10","arxiv_id":"2206.05266","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/does-self-supervised-learning-really-improve#ran","syntology_url":"https://syntology.ai/paper/2206.05266","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05266"}},"official":{"repos":["LostXine/elo-sac","lostxine/elo-rainbow"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/challenges-and-opportunities-in-offline","slug":"challenges-and-opportunities-in-offline","title":"Challenges and Opportunities in Offline Reinforcement Learning from Visual Observations","date":"2022-06-09","arxiv_id":"2206.04779","repositories_listed":2,"syntology":null},{"url":"/paper/on-reinforcement-learning-and-distribution","slug":"on-reinforcement-learning-and-distribution","title":"On Reinforcement Learning and Distribution Matching for Fine-Tuning Language Models with no Catastrophic Forgetting","date":"2022-06-01","arxiv_id":"2206.00761","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/on-reinforcement-learning-and-distribution#ran","syntology_url":"https://syntology.ai/paper/2206.00761","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.00761"}},"official":{"repos":["naver/gdc"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/task-agnostic-continual-reinforcement","slug":"task-agnostic-continual-reinforcement","title":"Task-Agnostic Continual Reinforcement Learning: Gaining Insights and Overcoming Challenges","date":"2022-05-28","arxiv_id":"2205.14495","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/task-agnostic-continual-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2205.14495","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14495"}},"official":{"repos":["amazon-science/replay-based-recurrent-rl","amazon-research/replay-based-recurrent-rl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/why-so-pessimistic-estimating-uncertainties-1","slug":"why-so-pessimistic-estimating-uncertainties-1","title":"Why So Pessimistic? Estimating Uncertainties for Offline RL through Ensembles, and Why Their Independence Matters","date":"2022-05-27","arxiv_id":"2205.13703","repositories_listed":2,"syntology":null},{"url":"/paper/history-compression-via-language-models-in","slug":"history-compression-via-language-models-in","title":"History Compression via Language Models in Reinforcement Learning","date":"2022-05-24","arxiv_id":"2205.12258","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/history-compression-via-language-models-in#ran","syntology_url":"https://syntology.ai/paper/2205.12258","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12258"}},"official":{"repos":["ml-jku/helm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-uncertainty-for-exploration-in-1","slug":"reward-uncertainty-for-exploration-in-1","title":"Reward Uncertainty for Exploration in Preference-based Reinforcement Learning","date":"2022-05-24","arxiv_id":"2205.12401","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reward-uncertainty-for-exploration-in-1#ran","syntology_url":"https://syntology.ai/paper/2205.12401","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12401"}},"official":null}},{"url":"/paper/distance-sensitive-offline-reinforcement","slug":"distance-sensitive-offline-reinforcement","title":"When Data Geometry Meets Deep Function: Generalizing Offline Reinforcement Learning","date":"2022-05-23","arxiv_id":"2205.11027","repositories_listed":2,"syntology":null},{"url":"/paper/reachability-constrained-reinforcement","slug":"reachability-constrained-reinforcement","title":"Reachability Constrained Reinforcement Learning","date":"2022-05-16","arxiv_id":"2205.07536","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":3,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reachability-constrained-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2205.07536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.07536"}},"official":{"repos":["mahaitongdae/Reachability_Constrained_RL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/efficient-risk-averse-reinforcement-learning","slug":"efficient-risk-averse-reinforcement-learning","title":"Efficient Risk-Averse Reinforcement Learning","date":"2022-05-10","arxiv_id":"2205.05138","repositories_listed":2,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/efficient-risk-averse-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2205.05138","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.05138"}},"official":{"repos":["ido90/CeSoR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/multivariate-prediction-intervals-for-random","slug":"multivariate-prediction-intervals-for-random","title":"Multivariate Prediction Intervals for Random Forests","date":"2022-05-04","arxiv_id":"2205.02260","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multivariate-prediction-intervals-for-random#ran","syntology_url":"https://syntology.ai/paper/2205.02260","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.02260"}},"official":{"repos":["CitrineInformatics/lolo","citrineinformatics/multivariate-prediction-intervals"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rambo-rl-robust-adversarial-model-based","slug":"rambo-rl-robust-adversarial-model-based","title":"RAMBO-RL: Robust Adversarial Model-Based Offline Reinforcement Learning","date":"2022-04-26","arxiv_id":"2204.12581","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rambo-rl-robust-adversarial-model-based#ran","syntology_url":"https://syntology.ai/paper/2204.12581","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.12581"}},"official":{"repos":["marc-rigter/rambo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chai-a-chatbot-ai-for-task-oriented-dialogue","slug":"chai-a-chatbot-ai-for-task-oriented-dialogue","title":"CHAI: A CHatbot AI for Task-Oriented Dialogue with Offline Reinforcement Learning","date":"2022-04-18","arxiv_id":"2204.08426","repositories_listed":2,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/chai-a-chatbot-ai-for-task-oriented-dialogue#ran","syntology_url":"https://syntology.ai/paper/2204.08426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.08426"}},"official":{"repos":["siddharthverma314/chai-naacl-2022"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-on-graph-a-survey","slug":"reinforcement-learning-on-graph-a-survey","title":"Reinforcement learning on graphs: A survey","date":"2022-04-13","arxiv_id":"2204.06127","repositories_listed":2,"syntology":null},{"url":"/paper/grounding-hindsight-instructions-in-multi","slug":"grounding-hindsight-instructions-in-multi","title":"Grounding Hindsight Instructions in Multi-Goal Reinforcement Learning for Robotics","date":"2022-04-08","arxiv_id":"2204.04308","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grounding-hindsight-instructions-in-multi#ran","syntology_url":"https://syntology.ai/paper/2204.04308","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.04308"}},"official":{"repos":["knowledgetechnologyuhh/hipss"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hysteresis-based-rl-robustifying","slug":"hysteresis-based-rl-robustifying","title":"Hysteresis-Based RL: Robustifying Reinforcement Learning-based Control Policies via Hybrid Control","date":"2022-04-01","arxiv_id":"2204.00654","repositories_listed":2,"syntology":null},{"url":"/paper/5g-routing-interfered-environment","slug":"5g-routing-interfered-environment","title":"5G Routing Interfered Environment","date":"2022-03-28","arxiv_id":"2203.14790","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-with-action-free-pre","slug":"reinforcement-learning-with-action-free-pre","title":"Reinforcement Learning with Action-Free Pre-Training from Videos","date":"2022-03-25","arxiv_id":"2203.13880","repositories_listed":2,"syntology":null},{"url":"/paper/reasoning-about-counterfactuals-to-improve","slug":"reasoning-about-counterfactuals-to-improve","title":"Reasoning about Counterfactuals to Improve Human Inverse Reinforcement Learning","date":"2022-03-03","arxiv_id":"2203.01855","repositories_listed":2,"syntology":null},{"url":"/paper/combining-modular-skills-in-multitask","slug":"combining-modular-skills-in-multitask","title":"Combining Modular Skills in Multitask Learning","date":"2022-02-28","arxiv_id":"2202.13914","repositories_listed":2,"syntology":null},{"url":"/paper/quantum-deep-reinforcement-learning-for-robot","slug":"quantum-deep-reinforcement-learning-for-robot","title":"Quantum Deep Reinforcement Learning for Robot Navigation Tasks","date":"2022-02-24","arxiv_id":"2202.12180","repositories_listed":2,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/quantum-deep-reinforcement-learning-for-robot#ran","syntology_url":"https://syntology.ai/paper/2202.12180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.12180"}},"official":{"repos":["dfki-ric-quantum/qdrl-turtlebot-env","dfki-ric-quantum/qdrl-turtlebot-eval"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/online-decision-transformer","slug":"online-decision-transformer","title":"Online Decision Transformer","date":"2022-02-11","arxiv_id":"2202.05607","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/online-decision-transformer#ran","syntology_url":"https://syntology.ai/paper/2202.05607","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.05607"}},"official":{"repos":["facebookresearch/online-dt"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"url":"/paper/smodice-versatile-offline-imitation-learning","slug":"smodice-versatile-offline-imitation-learning","title":"Versatile Offline Imitation from Observations and Examples via Regularized State-Occupancy Matching","date":"2022-02-04","arxiv_id":"2202.02433","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":4,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":4,"phrase":"6 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/smodice-versatile-offline-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/2202.02433","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.02433"}},"official":{"repos":["jasonma2016/smodice"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/constrained-variational-policy-optimization","slug":"constrained-variational-policy-optimization","title":"Constrained Variational Policy Optimization for Safe Reinforcement Learning","date":"2022-01-28","arxiv_id":"2201.11927","repositories_listed":2,"syntology":{"n":11,"n_ran":5,"n_constructed":2,"n_ran_checked":2,"n_instrument":3,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/constrained-variational-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2201.11927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.11927"}},"official":{"repos":["liuzuxin/cvpo-safe-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/simsr-simple-distance-based-state","slug":"simsr-simple-distance-based-state","title":"SimSR: Simple Distance-based State Representation for Deep Reinforcement Learning","date":"2021-12-31","arxiv_id":"2112.15303","repositories_listed":2,"syntology":null},{"url":"/paper/knowledge-intensive-state-design-for-traffic","slug":"knowledge-intensive-state-design-for-traffic","title":"Leveraging Queue Length and Attention Mechanisms for Enhanced Traffic Signal Control Optimization","date":"2021-12-30","arxiv_id":"2201.00006","repositories_listed":2,"syntology":null},{"url":"/paper/lane-change-decision-making-through-deep-1","slug":"lane-change-decision-making-through-deep-1","title":"Lane Change Decision-Making through Deep Reinforcement Learning","date":"2021-12-24","arxiv_id":"2112.14705","repositories_listed":2,"syntology":null},{"url":"/paper/on-the-unreasonable-efficiency-of-state-space","slug":"on-the-unreasonable-efficiency-of-state-space","title":"On the Unreasonable Efficiency of State Space Clustering in Personalization Tasks","date":"2021-12-24","arxiv_id":"2112.13141","repositories_listed":2,"syntology":null},{"url":"/paper/autonomous-reinforcement-learning-formalism-1","slug":"autonomous-reinforcement-learning-formalism-1","title":"Autonomous Reinforcement Learning: Formalism and Benchmarking","date":"2021-12-17","arxiv_id":"2112.09605","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autonomous-reinforcement-learning-formalism-1#ran","syntology_url":"https://syntology.ai/paper/2112.09605","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.09605"}},"official":{"repos":["architsharma97/earl_benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-share-in-multi-agent-1","slug":"learning-to-share-in-multi-agent-1","title":"Learning to Share in Multi-Agent Reinforcement Learning","date":"2021-12-16","arxiv_id":"2112.08702","repositories_listed":2,"syntology":null},{"url":"/paper/unsupervised-reinforcement-learning-in","slug":"unsupervised-reinforcement-learning-in","title":"Unsupervised Reinforcement Learning in Multiple Environments","date":"2021-12-16","arxiv_id":"2112.08746","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unsupervised-reinforcement-learning-in#ran","syntology_url":"https://syntology.ai/paper/2112.08746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.08746"}},"official":{"repos":["muttimirco/alphamepol"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-experimental-design-perspective-on-model","slug":"an-experimental-design-perspective-on-model","title":"An Experimental Design Perspective on Model-Based Reinforcement Learning","date":"2021-12-09","arxiv_id":"2112.05244","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":3,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-experimental-design-perspective-on-model#ran","syntology_url":"https://syntology.ai/paper/2112.05244","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.05244"}},"official":null}},{"url":"/paper/vmagent-scheduling-simulator-for","slug":"vmagent-scheduling-simulator-for","title":"VMAgent: Scheduling Simulator for Reinforcement Learning","date":"2021-12-09","arxiv_id":"2112.04785","repositories_listed":2,"syntology":null},{"url":"/paper/shinrl-a-library-for-evaluating-rl-algorithms","slug":"shinrl-a-library-for-evaluating-rl-algorithms","title":"ShinRL: A Library for Evaluating RL Algorithms from Theoretical and Practical Perspectives","date":"2021-12-08","arxiv_id":"2112.04123","repositories_listed":2,"syntology":null},{"url":"/paper/efficient-pressure-improving-efficiency-for","slug":"efficient-pressure-improving-efficiency-for","title":"Efficient Pressure: Improving efficiency for signalized intersections","date":"2021-12-04","arxiv_id":"2112.02336","repositories_listed":2,"syntology":null},{"url":"/paper/efficient-symptom-inquiring-and-diagnosis-via","slug":"efficient-symptom-inquiring-and-diagnosis-via","title":"Efficient Symptom Inquiring and Diagnosis via Adaptive Alignment of Reinforcement Learning and Classification","date":"2021-12-01","arxiv_id":"2112.00733","repositories_listed":2,"syntology":null},{"url":"/paper/episodic-multi-agent-reinforcement-learning-1","slug":"episodic-multi-agent-reinforcement-learning-1","title":"Episodic Multi-agent Reinforcement Learning with Curiosity-Driven Exploration","date":"2021-11-22","arxiv_id":"2111.11032","repositories_listed":2,"syntology":null},{"url":"/paper/cleanrl-high-quality-single-file","slug":"cleanrl-high-quality-single-file","title":"CleanRL: High-quality Single-file Implementations of Deep Reinforcement Learning Algorithms","date":"2021-11-16","arxiv_id":"2111.08819","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cleanrl-high-quality-single-file#ran","syntology_url":"https://syntology.ai/paper/2111.08819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.08819"}},"official":{"repos":["vwxyzjn/cleanrl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/understanding-the-effects-of-dataset","slug":"understanding-the-effects-of-dataset","title":"A Dataset Perspective on Offline Reinforcement Learning","date":"2021-11-08","arxiv_id":"2111.04714","repositories_listed":2,"syntology":null},{"url":"/paper/d3rlpy-an-offline-deep-reinforcement-learning","slug":"d3rlpy-an-offline-deep-reinforcement-learning","title":"d3rlpy: An Offline Deep Reinforcement Learning Library","date":"2021-11-06","arxiv_id":"2111.03788","repositories_listed":2,"syntology":null},{"url":"/paper/learning-multiresolution-matrix-factorization","slug":"learning-multiresolution-matrix-factorization","title":"Learning Multiresolution Matrix Factorization and its Wavelet Networks on Graphs","date":"2021-11-02","arxiv_id":"2111.01940","repositories_listed":2,"syntology":null},{"url":"/paper/context-meta-reinforcement-learning-via","slug":"context-meta-reinforcement-learning-via","title":"Context Meta-Reinforcement Learning via Neuromodulation","date":"2021-10-30","arxiv_id":"2111.00134","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/context-meta-reinforcement-learning-via#ran","syntology_url":"https://syntology.ai/paper/2111.00134","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.00134"}},"official":{"repos":["dlpbc/nm-metarl","soltoggio/ct-graph"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/intrusion-prevention-through-optimal-stopping","slug":"intrusion-prevention-through-optimal-stopping","title":"Intrusion Prevention through Optimal Stopping","date":"2021-10-30","arxiv_id":"2111.00289","repositories_listed":2,"syntology":null},{"url":"/paper/on-joint-learning-for-solving-placement-and","slug":"on-joint-learning-for-solving-placement-and","title":"On Joint Learning for Solving Placement and Routing in Chip Design","date":"2021-10-30","arxiv_id":"2111.00234","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-joint-learning-for-solving-placement-and#ran","syntology_url":"https://syntology.ai/paper/2111.00234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.00234"}},"official":{"repos":["thinklab-sjtu/eda-ai"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/distributional-reinforcement-learning-for-4","slug":"distributional-reinforcement-learning-for-4","title":"Distributional Reinforcement Learning for Multi-Dimensional Reward Functions","date":"2021-10-26","arxiv_id":"2110.13578","repositories_listed":2,"syntology":null},{"url":"/paper/fault-tolerant-federated-reinforcement","slug":"fault-tolerant-federated-reinforcement","title":"Fault-Tolerant Federated Reinforcement Learning with Theoretical Guarantee","date":"2021-10-26","arxiv_id":"2110.14074","repositories_listed":2,"syntology":{"n":15,"n_ran":9,"n_constructed":3,"n_ran_checked":4,"n_instrument":5,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":15,"phrase":"9 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/fault-tolerant-federated-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2110.14074","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.14074"}},"official":{"repos":["flint-xf-fan/Byzantine-Federeated-RL"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/model-a-modularized-end-to-end-reinforcement","slug":"model-a-modularized-end-to-end-reinforcement","title":"A Versatile and Efficient Reinforcement Learning Framework for Autonomous Driving","date":"2021-10-22","arxiv_id":"2110.11573","repositories_listed":2,"syntology":null},{"url":"/paper/cora-benchmarks-baselines-and-metrics-as-a","slug":"cora-benchmarks-baselines-and-metrics-as-a","title":"CORA: Benchmarks, Baselines, and Metrics as a Platform for Continual Reinforcement Learning Agents","date":"2021-10-19","arxiv_id":"2110.10067","repositories_listed":2,"syntology":null},{"url":"/paper/learning-temporally-consistent-1","slug":"learning-temporally-consistent-1","title":"Learning Temporally-Consistent Representations for Data-Efficient Reinforcement Learning","date":"2021-10-11","arxiv_id":"2110.04935","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-temporally-consistent-1#ran","syntology_url":"https://syntology.ai/paper/2110.04935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.04935"}},"official":{"repos":["anon-researcher-repo/ksl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dropout-q-functions-for-doubly-efficient","slug":"dropout-q-functions-for-doubly-efficient","title":"Dropout Q-Functions for Doubly Efficient Reinforcement Learning","date":"2021-10-05","arxiv_id":"2110.02034","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":3,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dropout-q-functions-for-doubly-efficient#ran","syntology_url":"https://syntology.ai/paper/2110.02034","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.02034"}},"official":{"repos":["TakuyaHiraoka/Dropout-Q-Functions-for-Doubly-Efficient-Reinforcement-Learning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}}],"record_sha256":"ab7f30e7e2096b07bedd7dba542d182b1ab972279cdf038182e1efde7497c41c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}