{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/4","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":152,"rows_per_page":100,"rows":[301,400],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/3","next":"/task/reinforcement-learning-1/papers/5","papers":[{"url":"/paper/reward-machines-exploiting-reward-function","slug":"reward-machines-exploiting-reward-function","title":"Reward Machines: Exploiting Reward Function Structure in Reinforcement Learning","date":"2020-10-06","arxiv_id":"2010.03950","repositories_listed":3,"syntology":null},{"url":"/paper/action-guidance-getting-the-best-of-sparse-1","slug":"action-guidance-getting-the-best-of-sparse-1","title":"Action Guidance: Getting the Best of Sparse Rewards and Shaped Rewards for Real-time Strategy Games","date":"2020-10-05","arxiv_id":"2010.03956","repositories_listed":3,"syntology":null},{"url":"/paper/decoupling-representation-learning-from","slug":"decoupling-representation-learning-from","title":"Decoupling Representation Learning from Reinforcement Learning","date":"2020-09-14","arxiv_id":"2009.08319","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/decoupling-representation-learning-from#ran","syntology_url":"https://syntology.ai/paper/2009.08319","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.08319"}},"official":{"repos":["astooke/rlpyt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/phasic-policy-gradient","slug":"phasic-policy-gradient","title":"Phasic Policy Gradient","date":"2020-09-09","arxiv_id":"2009.04416","repositories_listed":3,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/phasic-policy-gradient#ran","syntology_url":"https://syntology.ai/paper/2009.04416","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.04416"}},"official":{"repos":["openai/phasic-policy-gradient"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/flightmare-a-flexible-quadrotor-simulator","slug":"flightmare-a-flexible-quadrotor-simulator","title":"Flightmare: A Flexible Quadrotor Simulator","date":"2020-09-01","arxiv_id":"2009.00563","repositories_listed":3,"syntology":null},{"url":"/paper/or-gym-a-reinforcement-learning-library-for","slug":"or-gym-a-reinforcement-learning-library-for","title":"OR-Gym: A Reinforcement Learning Library for Operations Research Problems","date":"2020-08-14","arxiv_id":"2008.06319","repositories_listed":3,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/or-gym-a-reinforcement-learning-library-for#ran","syntology_url":"https://syntology.ai/paper/2008.06319","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.06319"}},"official":{"repos":["hubbs5/or-gym"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/aligning-ai-with-shared-human-values","slug":"aligning-ai-with-shared-human-values","title":"Aligning AI With Shared Human Values","date":"2020-08-05","arxiv_id":"2008.02275","repositories_listed":3,"syntology":{"n":7,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/aligning-ai-with-shared-human-values#ran","syntology_url":"https://syntology.ai/paper/2008.02275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.02275"}},"official":{"repos":["hendrycks/ethics"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/babyai-1-1","slug":"babyai-1-1","title":"BabyAI 1.1","date":"2020-07-24","arxiv_id":"2007.12770","repositories_listed":3,"syntology":{"n":9,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/babyai-1-1#ran","syntology_url":"https://syntology.ai/paper/2007.12770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.12770"}},"official":{"repos":["mila-iqia/babyai"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/monte-carlo-tree-search-as-regularized-policy","slug":"monte-carlo-tree-search-as-regularized-policy","title":"Monte-Carlo Tree Search as Regularized Policy Optimization","date":"2020-07-24","arxiv_id":"2007.12509","repositories_listed":3,"syntology":null},{"url":"/paper/implicit-distributional-reinforcement","slug":"implicit-distributional-reinforcement","title":"Implicit Distributional Reinforcement Learning","date":"2020-07-13","arxiv_id":"2007.06159","repositories_listed":3,"syntology":{"n":6,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/implicit-distributional-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2007.06159","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.06159"}},"official":{"repos":["zhougroup/IDAC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/uav-path-planning-for-wireless-data","slug":"uav-path-planning-for-wireless-data","title":"UAV Path Planning for Wireless Data Harvesting: A Deep Reinforcement Learning Approach","date":"2020-07-01","arxiv_id":"2007.00544","repositories_listed":3,"syntology":null},{"url":"/paper/the-nethack-learning-environment","slug":"the-nethack-learning-environment","title":"The NetHack Learning Environment","date":"2020-06-24","arxiv_id":"2006.13760","repositories_listed":3,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/the-nethack-learning-environment#ran","syntology_url":"https://syntology.ai/paper/2006.13760","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.13760"}},"official":{"repos":["facebookresearch/nle"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/adversarial-soft-advantage-fitting-imitation","slug":"adversarial-soft-advantage-fitting-imitation","title":"Adversarial Soft Advantage Fitting: Imitation Learning without Policy Optimization","date":"2020-06-23","arxiv_id":"2006.13258","repositories_listed":3,"syntology":null},{"url":"/paper/shared-experience-actor-critic-for-multi","slug":"shared-experience-actor-critic-for-multi","title":"Shared Experience Actor-Critic for Multi-Agent Reinforcement Learning","date":"2020-06-12","arxiv_id":"2006.07169","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-real","slug":"deep-reinforcement-learning-for-real","title":"Deep Reinforcement learning for real autonomous mobile robot navigation in indoor environments","date":"2020-05-28","arxiv_id":"2005.13857","repositories_listed":3,"syntology":null},{"url":"/paper/implementation-matters-in-deep-policy","slug":"implementation-matters-in-deep-policy","title":"Implementation Matters in Deep Policy Gradients: A Case Study on PPO and TRPO","date":"2020-05-25","arxiv_id":"2005.12729","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/implementation-matters-in-deep-policy#ran","syntology_url":"https://syntology.ai/paper/2005.12729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.12729"}},"official":{"repos":["MadryLab/implementation-matters"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-reinforcement-learning-tutorial","slug":"offline-reinforcement-learning-tutorial","title":"Offline Reinforcement Learning: Tutorial, Review, and Perspectives on Open Problems","date":"2020-05-04","arxiv_id":"2005.01643","repositories_listed":3,"syntology":null},{"url":"/paper/ultrasound-guided-robotic-navigation-with","slug":"ultrasound-guided-robotic-navigation-with","title":"Ultrasound-Guided Robotic Navigation with Deep Reinforcement Learning","date":"2020-03-30","arxiv_id":"2003.13321","repositories_listed":3,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ultrasound-guided-robotic-navigation-with#ran","syntology_url":"https://syntology.ai/paper/2003.13321","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.13321"}},"official":{"repos":["hhase/spinal-navigation-rl","hhase/sacrum_data-set"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/neuroevolution-of-self-interpretable-agents","slug":"neuroevolution-of-self-interpretable-agents","title":"Neuroevolution of Self-Interpretable Agents","date":"2020-03-18","arxiv_id":"2003.08165","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/neuroevolution-of-self-interpretable-agents#ran","syntology_url":"https://syntology.ai/paper/2003.08165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.08165"}},"official":null}},{"url":"/paper/sample-efficient-reinforcement-learning","slug":"sample-efficient-reinforcement-learning","title":"Sample Efficient Reinforcement Learning through Learning from Demonstrations in Minecraft","date":"2020-03-12","arxiv_id":"2003.06066","repositories_listed":3,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/sample-efficient-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2003.06066","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.06066"}},"official":null}},{"url":"/paper/tadam-a-robust-stochastic-gradient-optimizer","slug":"tadam-a-robust-stochastic-gradient-optimizer","title":"TAdam: A Robust Stochastic Gradient Optimizer","date":"2020-02-29","arxiv_id":"2003.00179","repositories_listed":3,"syntology":null},{"url":"/paper/a-deep-reinforcement-learning-algorithm-using","slug":"a-deep-reinforcement-learning-algorithm-using","title":"A Deep Reinforcement Learning Algorithm Using Dynamic Attention Model for Vehicle Routing Problems","date":"2020-02-09","arxiv_id":"2002.03282","repositories_listed":3,"syntology":null},{"url":"/paper/addressing-value-estimation-errors-in","slug":"addressing-value-estimation-errors-in","title":"Distributional Soft Actor-Critic: Off-Policy Reinforcement Learning for Addressing Value Estimation Errors","date":"2020-01-09","arxiv_id":"2001.02811","repositories_listed":3,"syntology":null},{"url":"/paper/imitation-learning-via-off-policy-1","slug":"imitation-learning-via-off-policy-1","title":"Imitation Learning via Off-Policy Distribution Matching","date":"2019-12-10","arxiv_id":"1912.05032","repositories_listed":3,"syntology":null},{"url":"/paper/efficient-object-detection-in-large-images","slug":"efficient-object-detection-in-large-images","title":"Efficient Object Detection in Large Images using Deep Reinforcement Learning","date":"2019-12-09","arxiv_id":"1912.03966","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-object-detection-in-large-images#ran","syntology_url":"https://syntology.ai/paper/1912.03966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.03966"}},"official":{"repos":["uzkent/EfficientObjectDetection"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/reinforcement-learning-upside-down-dont","slug":"reinforcement-learning-upside-down-dont","title":"Reinforcement Learning Upside Down: Don't Predict Rewards -- Just Map Them to Actions","date":"2019-12-05","arxiv_id":"1912.02875","repositories_listed":3,"syntology":null},{"url":"/paper/empirical-study-of-off-policy-policy","slug":"empirical-study-of-off-policy-policy","title":"Empirical Study of Off-Policy Policy Evaluation for Reinforcement Learning","date":"2019-11-15","arxiv_id":"1911.06854","repositories_listed":3,"syntology":null},{"url":"/paper/real-time-reinforcement-learning","slug":"real-time-reinforcement-learning","title":"Real-Time Reinforcement Learning","date":"2019-11-11","arxiv_id":"1911.04448","repositories_listed":3,"syntology":{"n":16,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/real-time-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1911.04448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.04448"}},"official":{"repos":["rmst/rtrl","elementai/avenue"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a-model-based-reinforcement-learning-with","slug":"a-model-based-reinforcement-learning-with","title":"Model-Based Reinforcement Learning with Adversarial Training for Online Recommendation","date":"2019-11-10","arxiv_id":"1911.03845","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-model-based-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/1911.03845","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.03845"}},"official":{"repos":["JianGuanTHU/IRecGAN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-reinforcement-learn-for-neural","slug":"learning-to-reinforcement-learn-for-neural","title":"Learning to reinforcement learn for Neural Architecture Search","date":"2019-11-09","arxiv_id":"1911.03769","repositories_listed":3,"syntology":null},{"url":"/paper/collision-avoidance-in-pedestrian-rich","slug":"collision-avoidance-in-pedestrian-rich","title":"Collision Avoidance in Pedestrian-Rich Environments with Deep Reinforcement Learning","date":"2019-10-24","arxiv_id":"1910.11689","repositories_listed":3,"syntology":null},{"url":"/paper/torchbeast-a-pytorch-platform-for-distributed","slug":"torchbeast-a-pytorch-platform-for-distributed","title":"TorchBeast: A PyTorch Platform for Distributed RL","date":"2019-10-08","arxiv_id":"1910.03552","repositories_listed":3,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/torchbeast-a-pytorch-platform-for-distributed#ran","syntology_url":"https://syntology.ai/paper/1910.03552","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.03552"}},"official":{"repos":["heiner/scalable_agent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/generalized-inner-loop-meta-learning","slug":"generalized-inner-loop-meta-learning","title":"Generalized Inner Loop Meta-Learning","date":"2019-10-03","arxiv_id":"1910.01727","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/generalized-inner-loop-meta-learning#ran","syntology_url":"https://syntology.ai/paper/1910.01727","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.01727"}},"official":{"repos":["learnables/learn2learn","facebookresearch/higher"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/reducing-overestimation-bias-in-multi-agent","slug":"reducing-overestimation-bias-in-multi-agent","title":"Reducing Overestimation Bias in Multi-Agent Domains Using Double Centralized Critics","date":"2019-10-03","arxiv_id":"1910.01465","repositories_listed":3,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reducing-overestimation-bias-in-multi-agent#ran","syntology_url":"https://syntology.ai/paper/1910.01465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.01465"}},"official":{"repos":["JohannesAck/MATD3implementation"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/c-3po-cyclic-three-phase-optimization-for","slug":"c-3po-cyclic-three-phase-optimization-for","title":"C-3PO: Cyclic-Three-Phase Optimization for Human-Robot Motion Retargeting based on Reinforcement Learning","date":"2019-09-25","arxiv_id":"1909.11303","repositories_listed":3,"syntology":null},{"url":"/paper/emergent-tool-use-from-multi-agent","slug":"emergent-tool-use-from-multi-agent","title":"Emergent Tool Use From Multi-Agent Autocurricula","date":"2019-09-17","arxiv_id":"1909.07528","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emergent-tool-use-from-multi-agent#ran","syntology_url":"https://syntology.ai/paper/1909.07528","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.07528"}},"official":{"repos":["openai/multi-agent-emergence-environments"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/behaviour-suite-for-reinforcement-learning","slug":"behaviour-suite-for-reinforcement-learning","title":"Behaviour Suite for Reinforcement Learning","date":"2019-08-09","arxiv_id":"1908.03568","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/behaviour-suite-for-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1908.03568","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.03568"}},"official":{"repos":["deepmind/bsuite"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/placeto-learning-generalizable-device","slug":"placeto-learning-generalizable-device","title":"Placeto: Learning Generalizable Device Placement Algorithms for Distributed Machine Learning","date":"2019-06-20","arxiv_id":"1906.08879","repositories_listed":3,"syntology":null},{"url":"/paper/boosting-soft-actor-critic-emphasizing-recent","slug":"boosting-soft-actor-critic-emphasizing-recent","title":"Boosting Soft Actor-Critic: Emphasizing Recent Experience without Forgetting the Past","date":"2019-06-10","arxiv_id":"1906.04009","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/boosting-soft-actor-critic-emphasizing-recent#ran","syntology_url":"https://syntology.ai/paper/1906.04009","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.04009"}},"official":null}},{"url":"/paper/reinforcement-learning-for-slate-based","slug":"reinforcement-learning-for-slate-based","title":"Reinforcement Learning for Slate-based Recommender Systems: A Tractable Decomposition and Practical Methodology","date":"2019-05-29","arxiv_id":"1905.12767","repositories_listed":3,"syntology":null},{"url":"/paper/maximum-entropy-regularized-multi-goal","slug":"maximum-entropy-regularized-multi-goal","title":"Maximum Entropy-Regularized Multi-Goal Reinforcement Learning","date":"2019-05-21","arxiv_id":"1905.08786","repositories_listed":3,"syntology":null},{"url":"/paper/recurrent-experience-replay-in-distributed","slug":"recurrent-experience-replay-in-distributed","title":"Recurrent Experience Replay in Distributed Reinforcement Learning","date":"2019-05-01","arxiv_id":null,"repositories_listed":3,"syntology":null},{"url":"/paper/end-to-end-robotic-reinforcement-learning","slug":"end-to-end-robotic-reinforcement-learning","title":"End-to-End Robotic Reinforcement Learning without Reward Engineering","date":"2019-04-16","arxiv_id":"1904.07854","repositories_listed":3,"syntology":null},{"url":"/paper/extrapolating-beyond-suboptimal","slug":"extrapolating-beyond-suboptimal","title":"Extrapolating Beyond Suboptimal Demonstrations via Inverse Reinforcement Learning from Observations","date":"2019-04-12","arxiv_id":"1904.06387","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/extrapolating-beyond-suboptimal#ran","syntology_url":"https://syntology.ai/paper/1904.06387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.06387"}},"official":{"repos":["hiwonjoon/ICML2019-TREX"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/holist-an-environment-for-machine-learning-of","slug":"holist-an-environment-for-machine-learning-of","title":"HOList: An Environment for Machine Learning of Higher-Order Theorem Proving","date":"2019-04-05","arxiv_id":"1904.03241","repositories_listed":3,"syntology":null},{"url":"/paper/improved-robustness-of-reinforcement-learning","slug":"improved-robustness-of-reinforcement-learning","title":"Improved robustness of reinforcement learning policies upon conversion to spiking neuronal network platforms applied to ATARI games","date":"2019-03-26","arxiv_id":"1903.11012","repositories_listed":3,"syntology":null},{"url":"/paper/minatar-an-atari-inspired-testbed-for-more","slug":"minatar-an-atari-inspired-testbed-for-more","title":"MinAtar: An Atari-Inspired Testbed for Thorough and Reproducible Reinforcement Learning Experiments","date":"2019-03-07","arxiv_id":"1903.03176","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/minatar-an-atari-inspired-testbed-for-more#ran","syntology_url":"https://syntology.ai/paper/1903.03176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.03176"}},"official":{"repos":["kenjyoung/MinAtar"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/hierarchical-critics-assignment-for-multi","slug":"hierarchical-critics-assignment-for-multi","title":"Reinforcement Learning from Hierarchical Critics","date":"2019-02-08","arxiv_id":"1902.03079","repositories_listed":3,"syntology":null},{"url":"/paper/pipps-flexible-model-based-policy-search","slug":"pipps-flexible-model-based-policy-search","title":"PIPPS: Flexible Model-Based Policy Search Robust to the Curse of Chaos","date":"2019-02-04","arxiv_id":"1902.01240","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-imbalanced","slug":"deep-reinforcement-learning-for-imbalanced","title":"Deep Reinforcement Learning for Imbalanced Classification","date":"2019-01-05","arxiv_id":"1901.01379","repositories_listed":3,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-for-imbalanced#ran","syntology_url":"https://syntology.ai/paper/1901.01379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.01379"}},"official":{"repos":["linenus/DRL-For-imbalanced-Classification"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/toybox-better-atari-environments-for-testing","slug":"toybox-better-atari-environments-for-testing","title":"ToyBox: Better Atari Environments for Testing Reinforcement Learning Agents","date":"2018-12-06","arxiv_id":"1812.02850","repositories_listed":3,"syntology":null},{"url":"/paper/scalable-agent-alignment-via-reward-modeling","slug":"scalable-agent-alignment-via-reward-modeling","title":"Scalable agent alignment via reward modeling: a research direction","date":"2018-11-19","arxiv_id":"1811.07871","repositories_listed":3,"syntology":null},{"url":"/paper/quota-the-quantile-option-architecture-for","slug":"quota-the-quantile-option-architecture-for","title":"QUOTA: The Quantile Option Architecture for Reinforcement Learning","date":"2018-11-05","arxiv_id":"1811.02073","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/quota-the-quantile-option-architecture-for#ran","syntology_url":"https://syntology.ai/paper/1811.02073","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.02073"}},"official":{"repos":["ShangtongZhang/DeepRL"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/social-influence-as-intrinsic-motivation-for","slug":"social-influence-as-intrinsic-motivation-for","title":"Social Influence as Intrinsic Motivation for Multi-Agent Deep Reinforcement Learning","date":"2018-10-19","arxiv_id":"1810.08647","repositories_listed":3,"syntology":null},{"url":"/paper/actor-attention-critic-for-multi-agent","slug":"actor-attention-critic-for-multi-agent","title":"Actor-Attention-Critic for Multi-Agent Reinforcement Learning","date":"2018-10-05","arxiv_id":"1810.02912","repositories_listed":3,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/actor-attention-critic-for-multi-agent#ran","syntology_url":"https://syntology.ai/paper/1810.02912","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.02912"}},"official":{"repos":["shariqiqbal2810/MAAC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/deep-quality-value-dqv-learning","slug":"deep-quality-value-dqv-learning","title":"Deep Quality-Value (DQV) Learning","date":"2018-09-30","arxiv_id":"1810.00368","repositories_listed":3,"syntology":null},{"url":"/paper/dynamic-weights-in-multi-objective-deep","slug":"dynamic-weights-in-multi-objective-deep","title":"Dynamic Weights in Multi-Objective Deep Reinforcement Learning","date":"2018-09-20","arxiv_id":"1809.07803","repositories_listed":3,"syntology":null},{"url":"/paper/multi-hop-knowledge-graph-reasoning-with","slug":"multi-hop-knowledge-graph-reasoning-with","title":"Multi-Hop Knowledge Graph Reasoning with Reward Shaping","date":"2018-08-31","arxiv_id":"1808.10568","repositories_listed":3,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-hop-knowledge-graph-reasoning-with#ran","syntology_url":"https://syntology.ai/paper/1808.10568","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.10568"}},"official":{"repos":["salesforce/MultiHopKG"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/decoupling-strategy-and-generation-in","slug":"decoupling-strategy-and-generation-in","title":"Decoupling Strategy and Generation in Negotiation Dialogues","date":"2018-08-29","arxiv_id":"1808.09637","repositories_listed":3,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/decoupling-strategy-and-generation-in#ran","syntology_url":"https://syntology.ai/paper/1808.09637","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.09637"}},"official":{"repos":["worksheets.codalab.org/worksheets/0x453913e76b65495d8b9730d41c7e0a0c"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/bipedal-walking-robot-using-deep","slug":"bipedal-walking-robot-using-deep","title":"Bipedal Walking Robot using Deep Deterministic Policy Gradient","date":"2018-07-16","arxiv_id":"1807.05924","repositories_listed":3,"syntology":null},{"url":"/paper/scheduled-policy-optimization-for-natural","slug":"scheduled-policy-optimization-for-natural","title":"Scheduled Policy Optimization for Natural Language Communication with Intelligent Agents","date":"2018-06-16","arxiv_id":"1806.06187","repositories_listed":3,"syntology":null},{"url":"/paper/surprising-negative-results-for-generative","slug":"surprising-negative-results-for-generative","title":"Surprising Negative Results for Generative Adversarial Tree Search","date":"2018-06-15","arxiv_id":"1806.05780","repositories_listed":3,"syntology":null},{"url":"/paper/maximum-a-posteriori-policy-optimisation","slug":"maximum-a-posteriori-policy-optimisation","title":"Maximum a Posteriori Policy Optimisation","date":"2018-06-14","arxiv_id":"1806.06920","repositories_listed":3,"syntology":null},{"url":"/paper/transfer-learning-for-related-reinforcement","slug":"transfer-learning-for-related-reinforcement","title":"Transfer Learning for Related Reinforcement Learning Tasks via Image-to-Image Translation","date":"2018-05-31","arxiv_id":"1806.07377","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-sequence-to","slug":"deep-reinforcement-learning-for-sequence-to","title":"Deep Reinforcement Learning For Sequence to Sequence Models","date":"2018-05-24","arxiv_id":"1805.09461","repositories_listed":3,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":15,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":2,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deep-reinforcement-learning-for-sequence-to#ran","syntology_url":"https://syntology.ai/paper/1805.09461","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.09461"}},"official":{"repos":["yaserkl/RLSeq2Seq"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/toward-diverse-text-generation-with-inverse","slug":"toward-diverse-text-generation-with-inverse","title":"Toward Diverse Text Generation with Inverse Reinforcement Learning","date":"2018-04-30","arxiv_id":"1804.11258","repositories_listed":3,"syntology":null},{"url":"/paper/gotta-learn-fast-a-new-benchmark-for","slug":"gotta-learn-fast-a-new-benchmark-for","title":"Gotta Learn Fast: A New Benchmark for Generalization in RL","date":"2018-04-10","arxiv_id":"1804.03720","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-traffic-light","slug":"deep-reinforcement-learning-for-traffic-light","title":"Deep Reinforcement Learning for Traffic Light Control in Vehicular Networks","date":"2018-03-29","arxiv_id":"1803.11115","repositories_listed":3,"syntology":null},{"url":"/paper/synthesizing-neural-network-controllers-with","slug":"synthesizing-neural-network-controllers-with","title":"Synthesizing Neural Network Controllers with Probabilistic Model based Reinforcement Learning","date":"2018-03-06","arxiv_id":"1803.02291","repositories_listed":3,"syntology":null},{"url":"/paper/mean-field-multi-agent-reinforcement-learning","slug":"mean-field-multi-agent-reinforcement-learning","title":"Mean Field Multi-Agent Reinforcement Learning","date":"2018-02-15","arxiv_id":"1802.05438","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mean-field-multi-agent-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1802.05438","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.05438"}},"official":{"repos":["mlii/mfrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/evolved-policy-gradients","slug":"evolved-policy-gradients","title":"Evolved Policy Gradients","date":"2018-02-13","arxiv_id":"1802.04821","repositories_listed":3,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evolved-policy-gradients#ran","syntology_url":"https://syntology.ai/paper/1802.04821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.04821"}},"official":{"repos":["openai/EPG"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/deep-dyna-q-integrating-planning-for-task","slug":"deep-dyna-q-integrating-planning-for-task","title":"Deep Dyna-Q: Integrating Planning for Task-Completion Dialogue Policy Learning","date":"2018-01-18","arxiv_id":"1801.06176","repositories_listed":3,"syntology":null},{"url":"/paper/jointly-learning-to-construct-and-control","slug":"jointly-learning-to-construct-and-control","title":"Jointly Learning to Construct and Control Agents using Deep Reinforcement Learning","date":"2018-01-04","arxiv_id":"1801.01432","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/jointly-learning-to-construct-and-control#ran","syntology_url":"https://syntology.ai/paper/1801.01432","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1801.01432"}},"official":null}},{"url":"/paper/rllib-abstractions-for-distributed","slug":"rllib-abstractions-for-distributed","title":"RLlib: Abstractions for Distributed Reinforcement Learning","date":"2017-12-26","arxiv_id":"1712.09381","repositories_listed":3,"syntology":null},{"url":"/paper/magent-a-many-agent-reinforcement-learning","slug":"magent-a-many-agent-reinforcement-learning","title":"MAgent: A Many-Agent Reinforcement Learning Platform for Artificial Collective Intelligence","date":"2017-12-02","arxiv_id":"1712.00600","repositories_listed":3,"syntology":null},{"url":"/paper/visualizing-and-understanding-atari-agents","slug":"visualizing-and-understanding-atari-agents","title":"Visualizing and Understanding Atari Agents","date":"2017-10-31","arxiv_id":"1711.00138","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visualizing-and-understanding-atari-agents#ran","syntology_url":"https://syntology.ai/paper/1711.00138","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1711.00138"}},"official":{"repos":["greydanus/visualize_atari"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/overcoming-exploration-in-reinforcement","slug":"overcoming-exploration-in-reinforcement","title":"Overcoming Exploration in Reinforcement Learning with Demonstrations","date":"2017-09-28","arxiv_id":"1709.10089","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/overcoming-exploration-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1709.10089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1709.10089"}},"official":null}},{"url":"/paper/a2-rl-aesthetics-aware-reinforcement-learning","slug":"a2-rl-aesthetics-aware-reinforcement-learning","title":"A2-RL: Aesthetics Aware Reinforcement Learning for Image Cropping","date":"2017-09-14","arxiv_id":"1709.04595","repositories_listed":3,"syntology":null},{"url":"/paper/chemgan-challenge-for-drug-discovery-can-ai","slug":"chemgan-challenge-for-drug-discovery-can-ai","title":"ChemGAN challenge for drug discovery: can AI reproduce natural chemical diversity?","date":"2017-08-28","arxiv_id":"1708.08227","repositories_listed":3,"syntology":null},{"url":"/paper/efficient-architecture-search-by-network","slug":"efficient-architecture-search-by-network","title":"Efficient Architecture Search by Network Transformation","date":"2017-07-16","arxiv_id":"1707.04873","repositories_listed":3,"syntology":null},{"url":"/paper/efficient-probabilistic-performance-bounds","slug":"efficient-probabilistic-performance-bounds","title":"Efficient Probabilistic Performance Bounds for Inverse Reinforcement Learning","date":"2017-07-03","arxiv_id":"1707.00724","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-probabilistic-performance-bounds#ran","syntology_url":"https://syntology.ai/paper/1707.00724","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1707.00724"}},"official":{"repos":["dsbrown1331/aaai-2018-code"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/reinforced-mnemonic-reader-for-machine","slug":"reinforced-mnemonic-reader-for-machine","title":"Reinforced Mnemonic Reader for Machine Reading Comprehension","date":"2017-05-08","arxiv_id":"1705.02798","repositories_listed":3,"syntology":null},{"url":"/paper/from-language-to-programs-bridging","slug":"from-language-to-programs-bridging","title":"From Language to Programs: Bridging Reinforcement Learning and Maximum Marginal Likelihood","date":"2017-04-25","arxiv_id":"1704.07926","repositories_listed":3,"syntology":null},{"url":"/paper/molecular-de-novo-design-through-deep","slug":"molecular-de-novo-design-through-deep","title":"Molecular De Novo Design through Deep Reinforcement Learning","date":"2017-04-25","arxiv_id":"1704.07555","repositories_listed":3,"syntology":null},{"url":"/paper/virtual-to-real-deep-reinforcement-learning","slug":"virtual-to-real-deep-reinforcement-learning","title":"Virtual-to-real Deep Reinforcement Learning: Continuous Control of Mobile Robots for Mapless Navigation","date":"2017-03-01","arxiv_id":"1703.00420","repositories_listed":3,"syntology":null},{"url":"/paper/hybrid-code-networks-practical-and-efficient","slug":"hybrid-code-networks-practical-and-efficient","title":"Hybrid Code Networks: practical and efficient end-to-end dialog control with supervised and reinforcement learning","date":"2017-02-10","arxiv_id":"1702.03274","repositories_listed":3,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hybrid-code-networks-practical-and-efficient#ran","syntology_url":"https://syntology.ai/paper/1702.03274","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1702.03274"}},"official":null}},{"url":"/paper/an-alternative-softmax-operator-for","slug":"an-alternative-softmax-operator-for","title":"An Alternative Softmax Operator for Reinforcement Learning","date":"2016-12-16","arxiv_id":"1612.05628","repositories_listed":3,"syntology":null},{"url":"/paper/cryptocurrency-portfolio-management-with-deep","slug":"cryptocurrency-portfolio-management-with-deep","title":"Cryptocurrency Portfolio Management with Deep Reinforcement Learning","date":"2016-12-05","arxiv_id":"1612.01277","repositories_listed":3,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cryptocurrency-portfolio-management-with-deep#ran","syntology_url":"https://syntology.ai/paper/1612.01277","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1612.01277"}},"official":{"repos":["ZhengyaoJiang/PGPortfolio"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/reinforcement-learning-through-asynchronous","slug":"reinforcement-learning-through-asynchronous","title":"Reinforcement Learning through Asynchronous Advantage Actor-Critic on a GPU","date":"2016-11-18","arxiv_id":"1611.06256","repositories_listed":3,"syntology":null},{"url":"/paper/reinforcement-learning-with-unsupervised","slug":"reinforcement-learning-with-unsupervised","title":"Reinforcement Learning with Unsupervised Auxiliary Tasks","date":"2016-11-16","arxiv_id":"1611.05397","repositories_listed":3,"syntology":null},{"url":"/paper/exploration-a-study-of-count-based","slug":"exploration-a-study-of-count-based","title":"#Exploration: A Study of Count-Based Exploration for Deep Reinforcement Learning","date":"2016-11-15","arxiv_id":"1611.04717","repositories_listed":3,"syntology":null},{"url":"/paper/a-connection-between-generative-adversarial","slug":"a-connection-between-generative-adversarial","title":"A Connection between Generative Adversarial Networks, Inverse Reinforcement Learning, and Energy-Based Models","date":"2016-11-11","arxiv_id":"1611.03852","repositories_listed":3,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/a-connection-between-generative-adversarial#ran","syntology_url":"https://syntology.ai/paper/1611.03852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1611.03852"}},"official":null}},{"url":"/paper/input-convex-neural-networks","slug":"input-convex-neural-networks","title":"Input Convex Neural Networks","date":"2016-09-22","arxiv_id":"1609.07152","repositories_listed":3,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/input-convex-neural-networks#ran","syntology_url":"https://syntology.ai/paper/1609.07152","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1609.07152"}},"official":{"repos":["locuslab/icnn"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-actor-critic-algorithm-for-sequence","slug":"an-actor-critic-algorithm-for-sequence","title":"An Actor-Critic Algorithm for Sequence Prediction","date":"2016-07-24","arxiv_id":"1607.07086","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/an-actor-critic-algorithm-for-sequence#ran","syntology_url":"https://syntology.ai/paper/1607.07086","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1607.07086"}},"official":{"repos":["rizar/actor-critic-public"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/model-free-episodic-control","slug":"model-free-episodic-control","title":"Model-Free Episodic Control","date":"2016-06-14","arxiv_id":"1606.04460","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/model-free-episodic-control#ran","syntology_url":"https://syntology.ai/paper/1606.04460","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1606.04460"}},"official":null}},{"url":"/paper/safe-and-efficient-off-policy-reinforcement","slug":"safe-and-efficient-off-policy-reinforcement","title":"Safe and Efficient Off-Policy Reinforcement Learning","date":"2016-06-08","arxiv_id":"1606.02647","repositories_listed":3,"syntology":null},{"url":"/paper/learning-to-communicate-with-deep-multi-agent","slug":"learning-to-communicate-with-deep-multi-agent","title":"Learning to Communicate with Deep Multi-Agent Reinforcement Learning","date":"2016-05-21","arxiv_id":"1605.06676","repositories_listed":3,"syntology":null},{"url":"/paper/data-efficient-off-policy-policy-evaluation","slug":"data-efficient-off-policy-policy-evaluation","title":"Data-Efficient Off-Policy Policy Evaluation for Reinforcement Learning","date":"2016-04-04","arxiv_id":"1604.00923","repositories_listed":3,"syntology":null},{"url":"/paper/learning-to-compose-neural-networks-for","slug":"learning-to-compose-neural-networks-for","title":"Learning to Compose Neural Networks for Question Answering","date":"2016-01-07","arxiv_id":"1601.01705","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-compose-neural-networks-for#ran","syntology_url":"https://syntology.ai/paper/1601.01705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1601.01705"}},"official":{"repos":["jacobandreas/nmn2"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/taming-the-noise-in-reinforcement-learning","slug":"taming-the-noise-in-reinforcement-learning","title":"Taming the Noise in Reinforcement Learning via Soft Updates","date":"2015-12-28","arxiv_id":"1512.08562","repositories_listed":3,"syntology":null}],"record_sha256":"cf60054011a10d6ac80deda1ce2c644f1c51b4926814ae4901fa2879ee7e4586","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}