{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/2","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":135,"rows_per_page":100,"rows":[101,200],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2","next":"/task/reinforcement-learning-2/papers/3","papers":[{"url":"/paper/deep-reinforcement-learning-in-parameterized","slug":"deep-reinforcement-learning-in-parameterized","title":"Deep Reinforcement Learning in Parameterized Action Space","date":"2015-11-13","arxiv_id":"1511.04143","repositories_listed":7,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-in-parameterized#ran","syntology_url":"https://syntology.ai/paper/1511.04143","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1511.04143"}},"official":{"repos":["mhauskn/dqn-hfo"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/physics-based-deep-learning","slug":"physics-based-deep-learning","title":"Physics-based Deep Learning","date":"2021-09-11","arxiv_id":"2109.05237","repositories_listed":6,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/physics-based-deep-learning#ran","syntology_url":"https://syntology.ai/paper/2109.05237","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.05237"}},"official":{"repos":["thunil/Physics-Based-Deep-Learning","tum-pbs/PhiFlow","tum-pbs/diffusion-based-flow-prediction","tum-pbs/pbdl-dataset","tum-pbs/pbdl-book"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["named_in_paper","official"]}}},{"url":"/paper/self-timed-reinforcement-learning-using","slug":"self-timed-reinforcement-learning-using","title":"Self-timed Reinforcement Learning using Tsetlin Machine","date":"2021-09-02","arxiv_id":"2109.00846","repositories_listed":6,"syntology":null},{"url":"/paper/finrl-a-deep-reinforcement-learning-library","slug":"finrl-a-deep-reinforcement-learning-library","title":"FinRL: A Deep Reinforcement Learning Library for Automated Stock Trading in Quantitative Finance","date":"2020-11-19","arxiv_id":"2011.09607","repositories_listed":6,"syntology":null},{"url":"/paper/munchausen-reinforcement-learning","slug":"munchausen-reinforcement-learning","title":"Munchausen Reinforcement Learning","date":"2020-07-28","arxiv_id":"2007.14430","repositories_listed":6,"syntology":{"n":22,"n_ran":12,"n_constructed":11,"n_ran_checked":12,"n_instrument":0,"n_unverified":10,"n_honours":0,"n_violates":1,"n_no_contract":11,"n_pointer_only":0,"phrase":"12 ran (of which 11 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/munchausen-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2007.14430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.14430"}},"official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/accelerating-online-reinforcement-learning","slug":"accelerating-online-reinforcement-learning","title":"AWAC: Accelerating Online Reinforcement Learning with Offline Datasets","date":"2020-06-16","arxiv_id":"2006.09359","repositories_listed":6,"syntology":{"n":13,"n_ran":10,"n_constructed":9,"n_ran_checked":9,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":8,"phrase":"10 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/accelerating-online-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2006.09359","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.09359"}},"official":{"repos":["vitchyr/rlkit"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/pcgrl-procedural-content-generation-via","slug":"pcgrl-procedural-content-generation-via","title":"PCGRL: Procedural Content Generation via Reinforcement Learning","date":"2020-01-24","arxiv_id":"2001.09212","repositories_listed":6,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pcgrl-procedural-content-generation-via#ran","syntology_url":"https://syntology.ai/paper/2001.09212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.09212"}},"official":{"repos":["amidos2006/gym-pcgrl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/leveraging-procedural-generation-to-benchmark","slug":"leveraging-procedural-generation-to-benchmark","title":"Leveraging Procedural Generation to Benchmark Reinforcement Learning","date":"2019-12-03","arxiv_id":"1912.01588","repositories_listed":6,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":3,"n_honours":2,"n_violates":1,"n_no_contract":5,"n_pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 1 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/leveraging-procedural-generation-to-benchmark#ran","syntology_url":"https://syntology.ai/paper/1912.01588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.01588"}},"official":{"repos":["openai/procgen","openai/train-procgen"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/fully-parameterized-quantile-function-for","slug":"fully-parameterized-quantile-function-for","title":"Fully Parameterized Quantile Function for Distributional Reinforcement Learning","date":"2019-11-05","arxiv_id":"1911.02140","repositories_listed":6,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/fully-parameterized-quantile-function-for#ran","syntology_url":"https://syntology.ai/paper/1911.02140","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.02140"}},"official":null}},{"url":"/paper/advantage-weighted-regression-simple-and","slug":"advantage-weighted-regression-simple-and","title":"Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning","date":"2019-10-01","arxiv_id":"1910.00177","repositories_listed":6,"syntology":null},{"url":"/paper/unsupervised-learning-of-object-keypoints-for","slug":"unsupervised-learning-of-object-keypoints-for","title":"Unsupervised Learning of Object Keypoints for Perception and Control","date":"2019-06-19","arxiv_id":"1906.11883","repositories_listed":6,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unsupervised-learning-of-object-keypoints-for#ran","syntology_url":"https://syntology.ai/paper/1906.11883","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.11883"}},"official":{"repos":["deepmind/deepmind-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/stroke-based-artistic-rendering-agent-with","slug":"stroke-based-artistic-rendering-agent-with","title":"Learning to Paint With Model-based Deep Reinforcement Learning","date":"2019-03-11","arxiv_id":"1903.04411","repositories_listed":6,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/stroke-based-artistic-rendering-agent-with#ran","syntology_url":"https://syntology.ai/paper/1903.04411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.04411"}},"official":{"repos":["hzwer/ICCV2019-LearningToPaint"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/near-optimal-representation-learning-for","slug":"near-optimal-representation-learning-for","title":"Near-Optimal Representation Learning for Hierarchical Reinforcement Learning","date":"2018-10-02","arxiv_id":"1810.01257","repositories_listed":6,"syntology":null},{"url":"/paper/evolution-guided-policy-gradient-in","slug":"evolution-guided-policy-gradient-in","title":"Evolution-Guided Policy Gradient in Reinforcement Learning","date":"2018-05-21","arxiv_id":"1805.07917","repositories_listed":6,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/evolution-guided-policy-gradient-in#ran","syntology_url":"https://syntology.ai/paper/1805.07917","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.07917"}},"official":{"repos":["ShawK91/erl_paper_nips18"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/motion-planning-among-dynamic-decision-making","slug":"motion-planning-among-dynamic-decision-making","title":"Motion Planning Among Dynamic, Decision-Making Agents with Deep Reinforcement Learning","date":"2018-05-04","arxiv_id":"1805.01956","repositories_listed":6,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/motion-planning-among-dynamic-decision-making#ran","syntology_url":"https://syntology.ai/paper/1805.01956","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.01956"}},"official":{"repos":["mfe7/cadrl_ros"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/deepmimic-example-guided-deep-reinforcement","slug":"deepmimic-example-guided-deep-reinforcement","title":"DeepMimic: Example-Guided Deep Reinforcement Learning of Physics-Based Character Skills","date":"2018-04-08","arxiv_id":"1804.02717","repositories_listed":6,"syntology":null},{"url":"/paper/deeptraffic-crowdsourced-hyperparameter","slug":"deeptraffic-crowdsourced-hyperparameter","title":"DeepTraffic: Crowdsourced Hyperparameter Tuning of Deep Reinforcement Learning Systems for Multi-Agent Dense Traffic Navigation","date":"2018-01-09","arxiv_id":"1801.02805","repositories_listed":6,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-unsupervised","slug":"deep-reinforcement-learning-for-unsupervised","title":"Deep Reinforcement Learning for Unsupervised Video Summarization with Diversity-Representativeness Reward","date":"2017-12-29","arxiv_id":"1801.00054","repositories_listed":6,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-for-unsupervised#ran","syntology_url":"https://syntology.ai/paper/1801.00054","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1801.00054"}},"official":{"repos":["KaiyangZhou/vsumm-reinforce"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/stochastic-answer-networks-for-machine","slug":"stochastic-answer-networks-for-machine","title":"Stochastic Answer Networks for Machine Reading Comprehension","date":"2017-12-10","arxiv_id":"1712.03556","repositories_listed":6,"syntology":null},{"url":"/paper/emergence-of-locomotion-behaviours-in-rich","slug":"emergence-of-locomotion-behaviours-in-rich","title":"Emergence of Locomotion Behaviours in Rich Environments","date":"2017-07-07","arxiv_id":"1707.02286","repositories_listed":6,"syntology":null},{"url":"/paper/virtual-to-real-reinforcement-learning-for","slug":"virtual-to-real-reinforcement-learning-for","title":"Virtual to Real Reinforcement Learning for Autonomous Driving","date":"2017-04-13","arxiv_id":"1704.03952","repositories_listed":6,"syntology":null},{"url":"/paper/deep-exploration-via-bootstrapped-dqn","slug":"deep-exploration-via-bootstrapped-dqn","title":"Deep Exploration via Bootstrapped DQN","date":"2016-02-15","arxiv_id":"1602.04621","repositories_listed":6,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deep-exploration-via-bootstrapped-dqn#ran","syntology_url":"https://syntology.ai/paper/1602.04621","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1602.04621"}},"official":null}},{"url":"/paper/lima-less-is-more-for-alignment","slug":"lima-less-is-more-for-alignment","title":"LIMA: Less Is More for Alignment","date":"2023-05-18","arxiv_id":"2305.11206","repositories_listed":5,"syntology":null},{"url":"/paper/reflexion-language-agents-with-verbal","slug":"reflexion-language-agents-with-verbal","title":"Reflexion: Language Agents with Verbal Reinforcement Learning","date":"2023-03-20","arxiv_id":"2303.11366","repositories_listed":5,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reflexion-language-agents-with-verbal#ran","syntology_url":"https://syntology.ai/paper/2303.11366","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11366"}},"official":{"repos":["noahshinn024/reflexion"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/corl-research-oriented-deep-offline-1","slug":"corl-research-oriented-deep-offline-1","title":"CORL: Research-oriented Deep Offline Reinforcement Learning Library","date":"2022-10-13","arxiv_id":"2210.07105","repositories_listed":5,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/corl-research-oriented-deep-offline-1#ran","syntology_url":"https://syntology.ai/paper/2210.07105","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07105"}},"official":{"repos":["corl-team/CORL","hanjuku-kaso/awesome-offline-rl","tinkoff-ai/CORL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/uncertainty-based-offline-reinforcement","slug":"uncertainty-based-offline-reinforcement","title":"Uncertainty-Based Offline Reinforcement Learning with Diversified Q-Ensemble","date":"2021-10-04","arxiv_id":"2110.01548","repositories_listed":5,"syntology":{"n":21,"n_ran":13,"n_constructed":11,"n_ran_checked":12,"n_instrument":1,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":6,"phrase":"13 ran (of which 11 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/uncertainty-based-offline-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2110.01548","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.01548"}},"official":null}},{"url":"/paper/learning-to-walk-in-minutes-using-massively","slug":"learning-to-walk-in-minutes-using-massively","title":"Learning to Walk in Minutes Using Massively Parallel Deep Reinforcement Learning","date":"2021-09-24","arxiv_id":"2109.11978","repositories_listed":5,"syntology":null},{"url":"/paper/safe-learning-in-robotics-from-learning-based","slug":"safe-learning-in-robotics-from-learning-based","title":"Safe Learning in Robotics: From Learning-Based Control to Safe Reinforcement Learning","date":"2021-08-13","arxiv_id":"2108.06266","repositories_listed":5,"syntology":null},{"url":"/paper/towards-mental-time-travel-a-hierarchical","slug":"towards-mental-time-travel-a-hierarchical","title":"Towards mental time travel: a hierarchical memory for reinforcement learning agents","date":"2021-05-28","arxiv_id":"2105.14039","repositories_listed":5,"syntology":{"n":6,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/towards-mental-time-travel-a-hierarchical#ran","syntology_url":"https://syntology.ai/paper/2105.14039","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.14039"}},"official":{"repos":["deepmind/deepmind-research","deepmind/dm_fast_mapping"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/gym-m-rts-toward-affordable-full-game-real","slug":"gym-m-rts-toward-affordable-full-game-real","title":"Gym-$μ$RTS: Toward Affordable Full Game Real-time Strategy Games Research with Deep Reinforcement Learning","date":"2021-05-21","arxiv_id":"2105.13807","repositories_listed":5,"syntology":null},{"url":"/paper/evolving-reinforcement-learning-algorithms-1","slug":"evolving-reinforcement-learning-algorithms-1","title":"Evolving Reinforcement Learning Algorithms","date":"2021-01-08","arxiv_id":"2101.03958","repositories_listed":5,"syntology":null},{"url":"/paper/smarts-scalable-multi-agent-reinforcement","slug":"smarts-scalable-multi-agent-reinforcement","title":"SMARTS: Scalable Multi-Agent Reinforcement Learning Training School for Autonomous Driving","date":"2020-10-19","arxiv_id":"2010.09776","repositories_listed":5,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/smarts-scalable-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2010.09776","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09776"}},"official":{"repos":["huawei-noah/SMARTS"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/acme-a-research-framework-for-distributed","slug":"acme-a-research-framework-for-distributed","title":"Acme: A Research Framework for Distributed Reinforcement Learning","date":"2020-06-01","arxiv_id":"2006.00979","repositories_listed":5,"syntology":null},{"url":"/paper/stabilizing-transformers-for-reinforcement-1","slug":"stabilizing-transformers-for-reinforcement-1","title":"Stabilizing Transformers for Reinforcement Learning","date":"2019-10-13","arxiv_id":"1910.06764","repositories_listed":5,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stabilizing-transformers-for-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/1910.06764","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.06764"}},"official":null}},{"url":"/paper/a-simple-randomization-technique-for-1","slug":"a-simple-randomization-technique-for-1","title":"Network Randomization: A Simple Technique for Generalization in Deep Reinforcement Learning","date":"2019-10-11","arxiv_id":"1910.05396","repositories_listed":5,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-simple-randomization-technique-for-1#ran","syntology_url":"https://syntology.ai/paper/1910.05396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.05396"}},"official":{"repos":["pokaxpoka/netrand"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-batch-deep-reinforcement","slug":"benchmarking-batch-deep-reinforcement","title":"Benchmarking Batch Deep Reinforcement Learning Algorithms","date":"2019-10-03","arxiv_id":"1910.01708","repositories_listed":5,"syntology":null},{"url":"/paper/multi-agent-deep-reinforcement-learning-for-3","slug":"multi-agent-deep-reinforcement-learning-for-3","title":"Multi-Agent Deep Reinforcement Learning for Liquidation Strategy Analysis","date":"2019-06-24","arxiv_id":"1906.11046","repositories_listed":5,"syntology":null},{"url":"/paper/sqil-imitation-learning-via-regularized","slug":"sqil-imitation-learning-via-regularized","title":"SQIL: Imitation Learning via Reinforcement Learning with Sparse Rewards","date":"2019-05-27","arxiv_id":"1905.11108","repositories_listed":5,"syntology":null},{"url":"/paper/crossnorm-normalization-for-off-policy-td","slug":"crossnorm-normalization-for-off-policy-td","title":"CrossQ: Batch Normalization in Deep Reinforcement Learning for Greater Sample Efficiency and Simplicity","date":"2019-02-14","arxiv_id":"1902.05605","repositories_listed":5,"syntology":null},{"url":"/paper/decoupling-feature-extraction-from-policy","slug":"decoupling-feature-extraction-from-policy","title":"Decoupling feature extraction from policy learning: assessing benefits of state representation learning in goal based robotics","date":"2019-01-24","arxiv_id":"1901.08651","repositories_listed":5,"syntology":null},{"url":"/paper/an-introduction-to-deep-reinforcement","slug":"an-introduction-to-deep-reinforcement","title":"An Introduction to Deep Reinforcement Learning","date":"2018-11-30","arxiv_id":"1811.12560","repositories_listed":5,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based","slug":"deep-reinforcement-learning-based","title":"Deep Reinforcement Learning based Recommendation with Explicit User-Item Interactions Modeling","date":"2018-10-29","arxiv_id":"1810.12027","repositories_listed":5,"syntology":null},{"url":"/paper/deep-reinforcement-learning","slug":"deep-reinforcement-learning","title":"Deep Reinforcement Learning","date":"2018-10-15","arxiv_id":"1810.06339","repositories_listed":5,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1810.06339","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.06339"}},"official":{"repos":["deepmind/lab","google/dopamine","mrkulk/hierarchical-deep-RL","rllab/rllab"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/parametrized-deep-q-networks-learning","slug":"parametrized-deep-q-networks-learning","title":"Parametrized Deep Q-Networks Learning: Reinforcement Learning with Discrete-Continuous Hybrid Action Space","date":"2018-10-10","arxiv_id":"1810.06394","repositories_listed":5,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/parametrized-deep-q-networks-learning#ran","syntology_url":"https://syntology.ai/paper/1810.06394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.06394"}},"official":null}},{"url":"/paper/s-rl-toolbox-environments-datasets-and","slug":"s-rl-toolbox-environments-datasets-and","title":"S-RL Toolbox: Environments, Datasets and Evaluation Metrics for State Representation Learning","date":"2018-09-25","arxiv_id":"1809.09369","repositories_listed":5,"syntology":null},{"url":"/paper/adversarial-deep-reinforcement-learning-in","slug":"adversarial-deep-reinforcement-learning-in","title":"Adversarial Deep Reinforcement Learning in Portfolio Management","date":"2018-08-29","arxiv_id":"1808.09940","repositories_listed":5,"syntology":null},{"url":"/paper/reinforcement-learning-on-web-interfaces","slug":"reinforcement-learning-on-web-interfaces","title":"Reinforcement Learning on Web Interfaces Using Workflow-Guided Exploration","date":"2018-02-24","arxiv_id":"1802.08802","repositories_listed":5,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reinforcement-learning-on-web-interfaces#ran","syntology_url":"https://syntology.ai/paper/1802.08802","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.08802"}},"official":{"repos":["stanfordnlp/wge","worksheets.codalab.org/worksheets/0x0f25031bd42f4aabbc17625fe1484066"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/fully-decentralized-multi-agent-reinforcement","slug":"fully-decentralized-multi-agent-reinforcement","title":"Fully Decentralized Multi-Agent Reinforcement Learning with Networked Agents","date":"2018-02-23","arxiv_id":"1802.08757","repositories_listed":5,"syntology":null},{"url":"/paper/action-branching-architectures-for-deep","slug":"action-branching-architectures-for-deep","title":"Action Branching Architectures for Deep Reinforcement Learning","date":"2017-11-24","arxiv_id":"1711.08946","repositories_listed":5,"syntology":null},{"url":"/paper/deep-learning-based-numerical-methods-for","slug":"deep-learning-based-numerical-methods-for","title":"Deep learning-based numerical methods for high-dimensional parabolic partial differential equations and backward stochastic differential equations","date":"2017-06-15","arxiv_id":"1706.04702","repositories_listed":5,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/deep-learning-based-numerical-methods-for#ran","syntology_url":"https://syntology.ai/paper/1706.04702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1706.04702"}},"official":null}},{"url":"/paper/efficient-parallel-methods-for-deep","slug":"efficient-parallel-methods-for-deep","title":"Efficient Parallel Methods for Deep Reinforcement Learning","date":"2017-05-13","arxiv_id":"1705.04862","repositories_listed":5,"syntology":null},{"url":"/paper/stabilising-experience-replay-for-deep-multi","slug":"stabilising-experience-replay-for-deep-multi","title":"Stabilising Experience Replay for Deep Multi-Agent Reinforcement Learning","date":"2017-02-28","arxiv_id":"1702.08887","repositories_listed":5,"syntology":null},{"url":"/paper/designing-neural-network-architectures-using","slug":"designing-neural-network-architectures-using","title":"Designing Neural Network Architectures using Reinforcement Learning","date":"2016-11-07","arxiv_id":"1611.02167","repositories_listed":5,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/designing-neural-network-architectures-using#ran","syntology_url":"https://syntology.ai/paper/1611.02167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1611.02167"}},"official":null}},{"url":"/paper/multiple-object-recognition-with-visual","slug":"multiple-object-recognition-with-visual","title":"Multiple Object Recognition with Visual Attention","date":"2014-12-24","arxiv_id":"1412.7755","repositories_listed":5,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multiple-object-recognition-with-visual#ran","syntology_url":"https://syntology.ai/paper/1412.7755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1412.7755"}},"official":null}},{"url":"/paper/hierarchical-reinforcement-learning-with-the","slug":"hierarchical-reinforcement-learning-with-the","title":"Hierarchical Reinforcement Learning with the MAXQ Value Function Decomposition","date":"1999-05-21","arxiv_id":"cs/9905014","repositories_listed":5,"syntology":null},{"url":"/paper/logic-rl-unleashing-llm-reasoning-with-rule","slug":"logic-rl-unleashing-llm-reasoning-with-rule","title":"Logic-RL: Unleashing LLM Reasoning with Rule-Based Reinforcement Learning","date":"2025-02-20","arxiv_id":"2502.14768","repositories_listed":4,"syntology":{"n":23,"n_ran":15,"n_constructed":0,"n_ran_checked":13,"n_instrument":2,"n_unverified":8,"n_honours":1,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 0 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/logic-rl-unleashing-llm-reasoning-with-rule#ran","syntology_url":"https://syntology.ai/paper/2502.14768","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14768"}},"official":{"repos":["Unakar/Logic-RL"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":6,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/openrlhf-an-easy-to-use-scalable-and-high","slug":"openrlhf-an-easy-to-use-scalable-and-high","title":"OpenRLHF: An Easy-to-use, Scalable and High-performance RLHF Framework","date":"2024-05-20","arxiv_id":"2405.11143","repositories_listed":4,"syntology":null},{"url":"/paper/dynamic-datasets-and-market-environments-for","slug":"dynamic-datasets-and-market-environments-for","title":"Dynamic Datasets and Market Environments for Financial Reinforcement Learning","date":"2023-04-25","arxiv_id":"2304.13174","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-datasets-and-market-environments-for#ran","syntology_url":"https://syntology.ai/paper/2304.13174","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.13174"}},"official":{"repos":["AI4Finance-Foundation/FinRL","ai4finance-foundation/finrl-meta"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-in-sample-softmax-for-offline","slug":"the-in-sample-softmax-for-offline","title":"The In-Sample Softmax for Offline Reinforcement Learning","date":"2023-02-28","arxiv_id":"2302.14372","repositories_listed":4,"syntology":null},{"url":"/paper/lifelong-reinforcement-learning-with","slug":"lifelong-reinforcement-learning-with","title":"Lifelong Reinforcement Learning with Modulating Masks","date":"2022-12-21","arxiv_id":"2212.11110","repositories_listed":4,"syntology":null},{"url":"/paper/finrl-meta-market-environments-and-benchmarks","slug":"finrl-meta-market-environments-and-benchmarks","title":"FinRL-Meta: Market Environments and Benchmarks for Data-Driven Financial Reinforcement Learning","date":"2022-11-06","arxiv_id":"2211.03107","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/finrl-meta-market-environments-and-benchmarks#ran","syntology_url":"https://syntology.ai/paper/2211.03107","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.03107"}},"official":{"repos":["AI4Finance-Foundation/FinRL","ai4finance-foundation/finrl-meta"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/training-a-helpful-and-harmless-assistant","slug":"training-a-helpful-and-harmless-assistant","title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","date":"2022-04-12","arxiv_id":"2204.05862","repositories_listed":4,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/training-a-helpful-and-harmless-assistant#ran","syntology_url":"https://syntology.ai/paper/2204.05862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.05862"}},"official":{"repos":["anthropics/hh-rlhf"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/decom-decomposed-policy-for-constrained","slug":"decom-decomposed-policy-for-constrained","title":"DeCOM: Decomposed Policy for Constrained Cooperative Multi-Agent Reinforcement Learning","date":"2021-11-10","arxiv_id":"2111.05670","repositories_listed":4,"syntology":null},{"url":"/paper/multi-agent-constrained-policy-optimisation","slug":"multi-agent-constrained-policy-optimisation","title":"Multi-Agent Constrained Policy Optimisation","date":"2021-10-06","arxiv_id":"2110.02793","repositories_listed":4,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/multi-agent-constrained-policy-optimisation#ran","syntology_url":"https://syntology.ai/paper/2110.02793","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.02793"}},"official":{"repos":["chauncygu/multi-agent-constrained-policy-optimisation"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/safe-control-gym-a-unified-benchmark-suite","slug":"safe-control-gym-a-unified-benchmark-suite","title":"safe-control-gym: a Unified Benchmark Suite for Safe Learning-based Control and Reinforcement Learning in Robotics","date":"2021-09-13","arxiv_id":"2109.06325","repositories_listed":4,"syntology":{"n":20,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/safe-control-gym-a-unified-benchmark-suite#ran","syntology_url":"https://syntology.ai/paper/2109.06325","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.06325"}},"official":{"repos":["utiasDSL/safe-control-gym"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/efficient-active-search-for-combinatorial","slug":"efficient-active-search-for-combinatorial","title":"Efficient Active Search for Combinatorial Optimization Problems","date":"2021-06-09","arxiv_id":"2106.05126","repositories_listed":4,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/efficient-active-search-for-combinatorial#ran","syntology_url":"https://syntology.ai/paper/2106.05126","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.05126"}},"official":{"repos":["ahottung/EAS"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/objective-robustness-in-deep-reinforcement","slug":"objective-robustness-in-deep-reinforcement","title":"Goal Misgeneralization in Deep Reinforcement Learning","date":"2021-05-28","arxiv_id":"2105.14111","repositories_listed":4,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/objective-robustness-in-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2105.14111","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.14111"}},"official":{"repos":["JacobPfau/procgenAISC","jbkjr/train-procgen-pytorch"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a-reinforcement-learning-environment-for-job","slug":"a-reinforcement-learning-environment-for-job","title":"A Reinforcement Learning Environment For Job-Shop Scheduling","date":"2021-04-08","arxiv_id":"2104.03760","repositories_listed":4,"syntology":null},{"url":"/paper/recurrent-rational-networks","slug":"recurrent-rational-networks","title":"Adaptive Rational Activations to Boost Deep Reinforcement Learning","date":"2021-02-18","arxiv_id":"2102.09407","repositories_listed":4,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recurrent-rational-networks#ran","syntology_url":"https://syntology.ai/paper/2102.09407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.09407"}},"official":{"repos":["ml-research/rational_activations","ml-research/rational_rl","ml-research/rational_sl","k4ntz/activation-functions"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-distracting-control-suite-a-challenging","slug":"the-distracting-control-suite-a-challenging","title":"The Distracting Control Suite -- A Challenging Benchmark for Reinforcement Learning from Pixels","date":"2021-01-07","arxiv_id":"2101.02722","repositories_listed":4,"syntology":null},{"url":"/paper/learning-to-dispatch-for-job-shop-scheduling","slug":"learning-to-dispatch-for-job-shop-scheduling","title":"Learning to Dispatch for Job Shop Scheduling via Deep Reinforcement Learning","date":"2020-10-23","arxiv_id":"2010.12367","repositories_listed":4,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-to-dispatch-for-job-shop-scheduling#ran","syntology_url":"https://syntology.ai/paper/2010.12367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.12367"}},"official":{"repos":["zcajiayin/L2D"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/d2rl-deep-dense-architectures-in-1","slug":"d2rl-deep-dense-architectures-in-1","title":"D2RL: Deep Dense Architectures in Reinforcement Learning","date":"2020-10-19","arxiv_id":"2010.09163","repositories_listed":4,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/d2rl-deep-dense-architectures-in-1#ran","syntology_url":"https://syntology.ai/paper/2010.09163","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09163"}},"official":{"repos":["pairlab/d2rl"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["found_in_text","listed","official"]}}},{"url":"/paper/htmrl-biologically-plausible-reinforcement","slug":"htmrl-biologically-plausible-reinforcement","title":"HTMRL: Biologically Plausible Reinforcement Learning with Hierarchical Temporal Memory","date":"2020-09-18","arxiv_id":"2009.08880","repositories_listed":4,"syntology":null},{"url":"/paper/meta-learning-through-hebbian-plasticity-in","slug":"meta-learning-through-hebbian-plasticity-in","title":"Meta-Learning through Hebbian Plasticity in Random Networks","date":"2020-07-06","arxiv_id":"2007.02686","repositories_listed":4,"syntology":null},{"url":"/paper/sample-factory-egocentric-3d-control-from","slug":"sample-factory-egocentric-3d-control-from","title":"Sample Factory: Egocentric 3D Control from Pixels at 100000 FPS with Asynchronous Reinforcement Learning","date":"2020-06-21","arxiv_id":"2006.11751","repositories_listed":4,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sample-factory-egocentric-3d-control-from#ran","syntology_url":"https://syntology.ai/paper/2006.11751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.11751"}},"official":{"repos":["alex-petrenko/sample-factory"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/weighted-qmix-expanding-monotonic-value","slug":"weighted-qmix-expanding-monotonic-value","title":"Weighted QMIX: Expanding Monotonic Value Function Factorisation for Deep Multi-Agent Reinforcement Learning","date":"2020-06-18","arxiv_id":"2006.10800","repositories_listed":4,"syntology":null},{"url":"/paper/deployment-efficient-reinforcement-learning","slug":"deployment-efficient-reinforcement-learning","title":"Deployment-Efficient Reinforcement Learning via Model-Based Offline Optimization","date":"2020-06-05","arxiv_id":"2006.03647","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deployment-efficient-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2006.03647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.03647"}},"official":{"repos":["matsuolab/BREMEN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generalized-state-dependent-exploration-for","slug":"generalized-state-dependent-exploration-for","title":"Smooth Exploration for Robotic Reinforcement Learning","date":"2020-05-12","arxiv_id":"2005.05719","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalized-state-dependent-exploration-for#ran","syntology_url":"https://syntology.ai/paper/2005.05719","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.05719"}},"official":{"repos":["DLR-RM/stable-baselines3"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/planning-to-explore-via-self-supervised-world","slug":"planning-to-explore-via-self-supervised-world","title":"Planning to Explore via Self-Supervised World Models","date":"2020-05-12","arxiv_id":"2005.05960","repositories_listed":4,"syntology":{"n":10,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/planning-to-explore-via-self-supervised-world#ran","syntology_url":"https://syntology.ai/paper/2005.05960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.05960"}},"official":{"repos":["ramanans1/plan2explore"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/task-oriented-dialogue-system-for-automatic-1","slug":"task-oriented-dialogue-system-for-automatic-1","title":"Hierarchical Reinforcement Learning for Automatic Disease Diagnosis","date":"2020-04-29","arxiv_id":"2004.14254","repositories_listed":4,"syntology":null},{"url":"/paper/image-augmentation-is-all-you-need","slug":"image-augmentation-is-all-you-need","title":"Image Augmentation Is All You Need: Regularizing Deep Reinforcement Learning from Pixels","date":"2020-04-28","arxiv_id":"2004.13649","repositories_listed":4,"syntology":{"n":10,"n_ran":6,"n_constructed":3,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":3,"n_pointer_only":8,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/image-augmentation-is-all-you-need#ran","syntology_url":"https://syntology.ai/paper/2004.13649","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.13649"}},"official":{"repos":["denisyarats/drq"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/robust-deep-reinforcement-learning-against","slug":"robust-deep-reinforcement-learning-against","title":"Robust Deep Reinforcement Learning against Adversarial Perturbations on State Observations","date":"2020-03-19","arxiv_id":"2003.08938","repositories_listed":4,"syntology":null},{"url":"/paper/discor-corrective-feedback-in-reinforcement","slug":"discor-corrective-feedback-in-reinforcement","title":"DisCor: Corrective Feedback in Reinforcement Learning via Distribution Correction","date":"2020-03-16","arxiv_id":"2003.07305","repositories_listed":4,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/discor-corrective-feedback-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2003.07305","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.07305"}},"official":null}},{"url":"/paper/interpretable-end-to-end-urban-autonomous","slug":"interpretable-end-to-end-urban-autonomous","title":"Interpretable End-to-end Urban Autonomous Driving with Latent Deep Reinforcement Learning","date":"2020-01-23","arxiv_id":"2001.08726","repositories_listed":4,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/interpretable-end-to-end-urban-autonomous#ran","syntology_url":"https://syntology.ai/paper/2001.08726","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.08726"}},"official":{"repos":["cjy1992/interp-e2e-driving"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/191202288","slug":"191202288","title":"Simplified Action Decoder for Deep Multi-Agent Reinforcement Learning","date":"2019-12-04","arxiv_id":"1912.02288","repositories_listed":4,"syntology":null},{"url":"/paper/comparing-observation-and-action","slug":"comparing-observation-and-action","title":"Comparing Observation and Action Representations for Deep Reinforcement Learning in $μ$RTS","date":"2019-10-26","arxiv_id":"1910.12134","repositories_listed":4,"syntology":null},{"url":"/paper/improving-sample-efficiency-in-model-free-1","slug":"improving-sample-efficiency-in-model-free-1","title":"Improving Sample Efficiency in Model-Free Reinforcement Learning from Images","date":"2019-10-02","arxiv_id":"1910.01741","repositories_listed":4,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-sample-efficiency-in-model-free-1#ran","syntology_url":"https://syntology.ai/paper/1910.01741","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.01741"}},"official":{"repos":["denisyarats/pytorch_sac_ae"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/sme-net-sparse-motion-estimation-for","slug":"sme-net-sparse-motion-estimation-for","title":"SME-Net: Sparse Motion Estimation for Parametric Video Prediction Through Reinforcement Learning","date":"2019-10-01","arxiv_id":null,"repositories_listed":4,"syntology":null},{"url":"/paper/a-generalized-algorithm-for-multi-objective","slug":"a-generalized-algorithm-for-multi-objective","title":"A Generalized Algorithm for Multi-Objective Reinforcement Learning and Policy Adaptation","date":"2019-08-21","arxiv_id":"1908.08342","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-generalized-algorithm-for-multi-objective#ran","syntology_url":"https://syntology.ai/paper/1908.08342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.08342"}},"official":{"repos":["RunzheYang/MORL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/collaborative-multi-agent-dialogue-model","slug":"collaborative-multi-agent-dialogue-model","title":"Collaborative Multi-Agent Dialogue Model Training Via Reinforcement Learning","date":"2019-07-11","arxiv_id":"1907.05507","repositories_listed":4,"syntology":null},{"url":"/paper/qtran-learning-to-factorize-with","slug":"qtran-learning-to-factorize-with","title":"QTRAN: Learning to Factorize with Transformation for Cooperative Multi-Agent Reinforcement Learning","date":"2019-05-14","arxiv_id":"1905.05408","repositories_listed":4,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/qtran-learning-to-factorize-with#ran","syntology_url":"https://syntology.ai/paper/1905.05408","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.05408"}},"official":{"repos":["Sonkyunghwan/QTRAN"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/multi-pass-q-networks-for-deep-reinforcement","slug":"multi-pass-q-networks-for-deep-reinforcement","title":"Multi-Pass Q-Networks for Deep Reinforcement Learning with Parameterised Action Spaces","date":"2019-05-10","arxiv_id":"1905.04388","repositories_listed":4,"syntology":null},{"url":"/paper/graph-convolutional-reinforcement-learning","slug":"graph-convolutional-reinforcement-learning","title":"Graph Convolutional Reinforcement Learning","date":"2018-10-22","arxiv_id":"1810.09202","repositories_listed":4,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/graph-convolutional-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1810.09202","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.09202"}},"official":{"repos":["PKU-AI-Edge/DGN"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/lift-reinforcement-learning-in-computer","slug":"lift-reinforcement-learning-in-computer","title":"LIFT: Reinforcement Learning in Computer Systems by Learning From Demonstrations","date":"2018-08-23","arxiv_id":"1808.07903","repositories_listed":4,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-playing-25d","slug":"deep-reinforcement-learning-for-playing-25d","title":"Deep Reinforcement Learning for Playing 2.5D Fighting Games","date":"2018-05-05","arxiv_id":"1805.02070","repositories_listed":4,"syntology":null},{"url":"/paper/learning-to-navigate-in-cities-without-a-map","slug":"learning-to-navigate-in-cities-without-a-map","title":"Learning to Navigate in Cities Without a Map","date":"2018-03-31","arxiv_id":"1804.00168","repositories_listed":4,"syntology":null},{"url":"/paper/learning-synergies-between-pushing-and","slug":"learning-synergies-between-pushing-and","title":"Learning Synergies between Pushing and Grasping with Self-supervised Deep Reinforcement Learning","date":"2018-03-27","arxiv_id":"1803.09956","repositories_listed":4,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-synergies-between-pushing-and#ran","syntology_url":"https://syntology.ai/paper/1803.09956","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1803.09956"}},"official":{"repos":["andyzeng/visual-pushing-grasping"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/deep-bayesian-bandits-showdown-an-empirical","slug":"deep-bayesian-bandits-showdown-an-empirical","title":"Deep Bayesian Bandits Showdown: An Empirical Comparison of Bayesian Deep Networks for Thompson Sampling","date":"2018-02-26","arxiv_id":"1802.09127","repositories_listed":4,"syntology":null},{"url":"/paper/diversity-is-all-you-need-learning-skills","slug":"diversity-is-all-you-need-learning-skills","title":"Diversity is All You Need: Learning Skills without a Reward Function","date":"2018-02-16","arxiv_id":"1802.06070","repositories_listed":4,"syntology":{"n":11,"n_ran":8,"n_constructed":5,"n_ran_checked":8,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 5 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/diversity-is-all-you-need-learning-skills#ran","syntology_url":"https://syntology.ai/paper/1802.06070","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.06070"}},"official":null}},{"url":"/paper/reinforcement-learning-for-solving-the","slug":"reinforcement-learning-for-solving-the","title":"Reinforcement Learning for Solving the Vehicle Routing Problem","date":"2018-02-12","arxiv_id":"1802.04240","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-for-solving-the#ran","syntology_url":"https://syntology.ai/paper/1802.04240","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.04240"}},"official":null}}],"record_sha256":"cf13b4a7288c9ec029df2c8fe17f2b72bf4fd821afb8a5c2205f5cdf1f6bc108","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}