{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/2","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":152,"rows_per_page":100,"rows":[101,200],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1","next":"/task/reinforcement-learning-1/papers/3","papers":[{"url":"/paper/vizdoom-competitions-playing-doom-from-pixels","slug":"vizdoom-competitions-playing-doom-from-pixels","title":"ViZDoom Competitions: Playing Doom from Pixels","date":"2018-09-10","arxiv_id":"1809.03470","repositories_listed":7,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/vizdoom-competitions-playing-doom-from-pixels#ran","syntology_url":"https://syntology.ai/paper/1809.03470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.03470"}},"official":{"repos":["mwydmuch/ViZDoom"],"state":"official: not harvested","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"url":"/paper/relational-deep-reinforcement-learning","slug":"relational-deep-reinforcement-learning","title":"Relational Deep Reinforcement Learning","date":"2018-06-05","arxiv_id":"1806.01830","repositories_listed":7,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/relational-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1806.01830","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.01830"}},"official":null}},{"url":"/paper/some-considerations-on-learning-to-explore","slug":"some-considerations-on-learning-to-explore","title":"Some Considerations on Learning to Explore via Meta-Reinforcement Learning","date":"2018-03-03","arxiv_id":"1803.01118","repositories_listed":7,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-list-wise","slug":"deep-reinforcement-learning-for-list-wise","title":"Deep Reinforcement Learning for List-wise Recommendations","date":"2017-12-30","arxiv_id":"1801.00209","repositories_listed":7,"syntology":null},{"url":"/paper/backpropagation-through-the-void-optimizing","slug":"backpropagation-through-the-void-optimizing","title":"Backpropagation through the Void: Optimizing control variates for black-box gradient estimation","date":"2017-10-31","arxiv_id":"1711.00123","repositories_listed":7,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":5,"n_honours":3,"n_violates":1,"n_no_contract":3,"n_pointer_only":12,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 3 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/backpropagation-through-the-void-optimizing#ran","syntology_url":"https://syntology.ai/paper/1711.00123","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1711.00123"}},"official":{"repos":["duvenaud/relax"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/learning-robust-rewards-with-adversarial","slug":"learning-robust-rewards-with-adversarial","title":"Learning Robust Rewards with Adversarial Inverse Reinforcement Learning","date":"2017-10-30","arxiv_id":"1710.11248","repositories_listed":7,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/learning-robust-rewards-with-adversarial#ran","syntology_url":"https://syntology.ai/paper/1710.11248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.11248"}},"official":null}},{"url":"/paper/deep-reinforcement-learning-from-human","slug":"deep-reinforcement-learning-from-human","title":"Deep reinforcement learning from human preferences","date":"2017-06-12","arxiv_id":"1706.03741","repositories_listed":7,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deep-reinforcement-learning-from-human#ran","syntology_url":"https://syntology.ai/paper/1706.03741","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1706.03741"}},"official":null}},{"url":"/paper/time-contrastive-networks-self-supervised","slug":"time-contrastive-networks-self-supervised","title":"Time-Contrastive Networks: Self-Supervised Learning from Video","date":"2017-04-23","arxiv_id":"1704.06888","repositories_listed":7,"syntology":null},{"url":"/paper/learning-cooperative-visual-dialog-agents","slug":"learning-cooperative-visual-dialog-agents","title":"Learning Cooperative Visual Dialog Agents with Deep Reinforcement Learning","date":"2017-03-20","arxiv_id":"1703.06585","repositories_listed":7,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-cooperative-visual-dialog-agents#ran","syntology_url":"https://syntology.ai/paper/1703.06585","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1703.06585"}},"official":null}},{"url":"/paper/robust-adversarial-reinforcement-learning","slug":"robust-adversarial-reinforcement-learning","title":"Robust Adversarial Reinforcement Learning","date":"2017-03-08","arxiv_id":"1703.02702","repositories_listed":7,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/robust-adversarial-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1703.02702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1703.02702"}},"official":null}},{"url":"/paper/sample-efficient-actor-critic-with-experience","slug":"sample-efficient-actor-critic-with-experience","title":"Sample Efficient Actor-Critic with Experience Replay","date":"2016-11-03","arxiv_id":"1611.01224","repositories_listed":7,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/sample-efficient-actor-critic-with-experience#ran","syntology_url":"https://syntology.ai/paper/1611.01224","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1611.01224"}},"official":null}},{"url":"/paper/playing-fps-games-with-deep-reinforcement","slug":"playing-fps-games-with-deep-reinforcement","title":"Playing FPS Games with Deep Reinforcement Learning","date":"2016-09-18","arxiv_id":"1609.05521","repositories_listed":7,"syntology":null},{"url":"/paper/deep-reinforcement-learning-from-self-play-in","slug":"deep-reinforcement-learning-from-self-play-in","title":"Deep Reinforcement Learning from Self-Play in Imperfect-Information Games","date":"2016-03-03","arxiv_id":"1603.01121","repositories_listed":7,"syntology":null},{"url":"/paper/deep-reinforcement-learning-in-parameterized","slug":"deep-reinforcement-learning-in-parameterized","title":"Deep Reinforcement Learning in Parameterized Action Space","date":"2015-11-13","arxiv_id":"1511.04143","repositories_listed":7,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-in-parameterized#ran","syntology_url":"https://syntology.ai/paper/1511.04143","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1511.04143"}},"official":{"repos":["mhauskn/dqn-hfo"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/physics-based-deep-learning","slug":"physics-based-deep-learning","title":"Physics-based Deep Learning","date":"2021-09-11","arxiv_id":"2109.05237","repositories_listed":6,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/physics-based-deep-learning#ran","syntology_url":"https://syntology.ai/paper/2109.05237","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.05237"}},"official":{"repos":["thunil/Physics-Based-Deep-Learning","tum-pbs/PhiFlow","tum-pbs/diffusion-based-flow-prediction","tum-pbs/pbdl-dataset","tum-pbs/pbdl-book"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["named_in_paper","official"]}}},{"url":"/paper/self-timed-reinforcement-learning-using","slug":"self-timed-reinforcement-learning-using","title":"Self-timed Reinforcement Learning using Tsetlin Machine","date":"2021-09-02","arxiv_id":"2109.00846","repositories_listed":6,"syntology":null},{"url":"/paper/habitat-2-0-training-home-assistants-to","slug":"habitat-2-0-training-home-assistants-to","title":"Habitat 2.0: Training Home Assistants to Rearrange their Habitat","date":"2021-06-28","arxiv_id":"2106.14405","repositories_listed":6,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/habitat-2-0-training-home-assistants-to#ran","syntology_url":"https://syntology.ai/paper/2106.14405","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.14405"}},"official":{"repos":["facebookresearch/habitat-lab"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/emergent-complexity-and-zero-shot-transfer-1","slug":"emergent-complexity-and-zero-shot-transfer-1","title":"Emergent Complexity and Zero-shot Transfer via Unsupervised Environment Design","date":"2020-12-03","arxiv_id":"2012.02096","repositories_listed":6,"syntology":null},{"url":"/paper/finrl-a-deep-reinforcement-learning-library","slug":"finrl-a-deep-reinforcement-learning-library","title":"FinRL: A Deep Reinforcement Learning Library for Automated Stock Trading in Quantitative Finance","date":"2020-11-19","arxiv_id":"2011.09607","repositories_listed":6,"syntology":null},{"url":"/paper/enhancing-graph-neural-network-based-fraud","slug":"enhancing-graph-neural-network-based-fraud","title":"Enhancing Graph Neural Network-based Fraud Detectors against Camouflaged Fraudsters","date":"2020-08-19","arxiv_id":"2008.08692","repositories_listed":6,"syntology":{"n":17,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/enhancing-graph-neural-network-based-fraud#ran","syntology_url":"https://syntology.ai/paper/2008.08692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.08692"}},"official":{"repos":["YingtongDou/CARE-GNN","safe-graph/DGFraud"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/munchausen-reinforcement-learning","slug":"munchausen-reinforcement-learning","title":"Munchausen Reinforcement Learning","date":"2020-07-28","arxiv_id":"2007.14430","repositories_listed":6,"syntology":{"n":22,"n_ran":12,"n_constructed":11,"n_ran_checked":12,"n_instrument":0,"n_unverified":10,"n_honours":0,"n_violates":1,"n_no_contract":11,"n_pointer_only":0,"phrase":"12 ran (of which 11 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/munchausen-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2007.14430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.14430"}},"official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/accelerating-online-reinforcement-learning","slug":"accelerating-online-reinforcement-learning","title":"AWAC: Accelerating Online Reinforcement Learning with Offline Datasets","date":"2020-06-16","arxiv_id":"2006.09359","repositories_listed":6,"syntology":{"n":13,"n_ran":10,"n_constructed":9,"n_ran_checked":9,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":8,"phrase":"10 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/accelerating-online-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2006.09359","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.09359"}},"official":{"repos":["vitchyr/rlkit"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/mopo-model-based-offline-policy-optimization","slug":"mopo-model-based-offline-policy-optimization","title":"MOPO: Model-based Offline Policy Optimization","date":"2020-05-27","arxiv_id":"2005.13239","repositories_listed":6,"syntology":{"n":8,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mopo-model-based-offline-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2005.13239","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.13239"}},"official":{"repos":["tianheyu927/mopo"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/pcgrl-procedural-content-generation-via","slug":"pcgrl-procedural-content-generation-via","title":"PCGRL: Procedural Content Generation via Reinforcement Learning","date":"2020-01-24","arxiv_id":"2001.09212","repositories_listed":6,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pcgrl-procedural-content-generation-via#ran","syntology_url":"https://syntology.ai/paper/2001.09212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.09212"}},"official":{"repos":["amidos2006/gym-pcgrl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/leveraging-procedural-generation-to-benchmark","slug":"leveraging-procedural-generation-to-benchmark","title":"Leveraging Procedural Generation to Benchmark Reinforcement Learning","date":"2019-12-03","arxiv_id":"1912.01588","repositories_listed":6,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":3,"n_honours":2,"n_violates":1,"n_no_contract":5,"n_pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 1 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/leveraging-procedural-generation-to-benchmark#ran","syntology_url":"https://syntology.ai/paper/1912.01588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.01588"}},"official":{"repos":["openai/procgen","openai/train-procgen"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/fully-parameterized-quantile-function-for","slug":"fully-parameterized-quantile-function-for","title":"Fully Parameterized Quantile Function for Distributional Reinforcement Learning","date":"2019-11-05","arxiv_id":"1911.02140","repositories_listed":6,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/fully-parameterized-quantile-function-for#ran","syntology_url":"https://syntology.ai/paper/1911.02140","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.02140"}},"official":null}},{"url":"/paper/advantage-weighted-regression-simple-and","slug":"advantage-weighted-regression-simple-and","title":"Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning","date":"2019-10-01","arxiv_id":"1910.00177","repositories_listed":6,"syntology":null},{"url":"/paper/unsupervised-learning-of-object-keypoints-for","slug":"unsupervised-learning-of-object-keypoints-for","title":"Unsupervised Learning of Object Keypoints for Perception and Control","date":"2019-06-19","arxiv_id":"1906.11883","repositories_listed":6,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unsupervised-learning-of-object-keypoints-for#ran","syntology_url":"https://syntology.ai/paper/1906.11883","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.11883"}},"official":{"repos":["deepmind/deepmind-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/stroke-based-artistic-rendering-agent-with","slug":"stroke-based-artistic-rendering-agent-with","title":"Learning to Paint With Model-based Deep Reinforcement Learning","date":"2019-03-11","arxiv_id":"1903.04411","repositories_listed":6,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/stroke-based-artistic-rendering-agent-with#ran","syntology_url":"https://syntology.ai/paper/1903.04411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.04411"}},"official":{"repos":["hzwer/ICCV2019-LearningToPaint"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/near-optimal-representation-learning-for","slug":"near-optimal-representation-learning-for","title":"Near-Optimal Representation Learning for Hierarchical Reinforcement Learning","date":"2018-10-02","arxiv_id":"1810.01257","repositories_listed":6,"syntology":null},{"url":"/paper/evolution-guided-policy-gradient-in","slug":"evolution-guided-policy-gradient-in","title":"Evolution-Guided Policy Gradient in Reinforcement Learning","date":"2018-05-21","arxiv_id":"1805.07917","repositories_listed":6,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/evolution-guided-policy-gradient-in#ran","syntology_url":"https://syntology.ai/paper/1805.07917","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.07917"}},"official":{"repos":["ShawK91/erl_paper_nips18"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/motion-planning-among-dynamic-decision-making","slug":"motion-planning-among-dynamic-decision-making","title":"Motion Planning Among Dynamic, Decision-Making Agents with Deep Reinforcement Learning","date":"2018-05-04","arxiv_id":"1805.01956","repositories_listed":6,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/motion-planning-among-dynamic-decision-making#ran","syntology_url":"https://syntology.ai/paper/1805.01956","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.01956"}},"official":{"repos":["mfe7/cadrl_ros"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/deepmimic-example-guided-deep-reinforcement","slug":"deepmimic-example-guided-deep-reinforcement","title":"DeepMimic: Example-Guided Deep Reinforcement Learning of Physics-Based Character Skills","date":"2018-04-08","arxiv_id":"1804.02717","repositories_listed":6,"syntology":null},{"url":"/paper/safe-exploration-in-continuous-action-spaces","slug":"safe-exploration-in-continuous-action-spaces","title":"Safe Exploration in Continuous Action Spaces","date":"2018-01-26","arxiv_id":"1801.08757","repositories_listed":6,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safe-exploration-in-continuous-action-spaces#ran","syntology_url":"https://syntology.ai/paper/1801.08757","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1801.08757"}},"official":null}},{"url":"/paper/deeptraffic-crowdsourced-hyperparameter","slug":"deeptraffic-crowdsourced-hyperparameter","title":"DeepTraffic: Crowdsourced Hyperparameter Tuning of Deep Reinforcement Learning Systems for Multi-Agent Dense Traffic Navigation","date":"2018-01-09","arxiv_id":"1801.02805","repositories_listed":6,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-unsupervised","slug":"deep-reinforcement-learning-for-unsupervised","title":"Deep Reinforcement Learning for Unsupervised Video Summarization with Diversity-Representativeness Reward","date":"2017-12-29","arxiv_id":"1801.00054","repositories_listed":6,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-for-unsupervised#ran","syntology_url":"https://syntology.ai/paper/1801.00054","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1801.00054"}},"official":{"repos":["KaiyangZhou/vsumm-reinforce"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/stochastic-answer-networks-for-machine","slug":"stochastic-answer-networks-for-machine","title":"Stochastic Answer Networks for Machine Reading Comprehension","date":"2017-12-10","arxiv_id":"1712.03556","repositories_listed":6,"syntology":null},{"url":"/paper/emergence-of-locomotion-behaviours-in-rich","slug":"emergence-of-locomotion-behaviours-in-rich","title":"Emergence of Locomotion Behaviours in Rich Environments","date":"2017-07-07","arxiv_id":"1707.02286","repositories_listed":6,"syntology":null},{"url":"/paper/virtual-to-real-reinforcement-learning-for","slug":"virtual-to-real-reinforcement-learning-for","title":"Virtual to Real Reinforcement Learning for Autonomous Driving","date":"2017-04-13","arxiv_id":"1704.03952","repositories_listed":6,"syntology":null},{"url":"/paper/deep-q-learning-from-demonstrations","slug":"deep-q-learning-from-demonstrations","title":"Deep Q-learning from Demonstrations","date":"2017-04-12","arxiv_id":"1704.03732","repositories_listed":6,"syntology":null},{"url":"/paper/deep-exploration-via-bootstrapped-dqn","slug":"deep-exploration-via-bootstrapped-dqn","title":"Deep Exploration via Bootstrapped DQN","date":"2016-02-15","arxiv_id":"1602.04621","repositories_listed":6,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deep-exploration-via-bootstrapped-dqn#ran","syntology_url":"https://syntology.ai/paper/1602.04621","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1602.04621"}},"official":null}},{"url":"/paper/r1-searcher-incentivizing-the-search","slug":"r1-searcher-incentivizing-the-search","title":"R1-Searcher: Incentivizing the Search Capability in LLMs via Reinforcement Learning","date":"2025-03-07","arxiv_id":"2503.05592","repositories_listed":5,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/r1-searcher-incentivizing-the-search#ran","syntology_url":"https://syntology.ai/paper/2503.05592","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.05592"}},"official":null}},{"url":"/paper/process-reinforcement-through-implicit","slug":"process-reinforcement-through-implicit","title":"Process Reinforcement through Implicit Rewards","date":"2025-02-03","arxiv_id":"2502.01456","repositories_listed":5,"syntology":{"n":17,"n_ran":14,"n_constructed":0,"n_ran_checked":13,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":2,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/process-reinforcement-through-implicit#ran","syntology_url":"https://syntology.ai/paper/2502.01456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.01456"}},"official":{"repos":["prime-rl/prime"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deepseek-v2-a-strong-economical-and-efficient","slug":"deepseek-v2-a-strong-economical-and-efficient","title":"DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model","date":"2024-05-07","arxiv_id":"2405.04434","repositories_listed":5,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deepseek-v2-a-strong-economical-and-efficient#ran","syntology_url":"https://syntology.ai/paper/2405.04434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.04434"}},"official":{"repos":["deepseek-ai/deepseek-v2"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/uncertainty-based-offline-reinforcement","slug":"uncertainty-based-offline-reinforcement","title":"Uncertainty-Based Offline Reinforcement Learning with Diversified Q-Ensemble","date":"2021-10-04","arxiv_id":"2110.01548","repositories_listed":5,"syntology":{"n":21,"n_ran":13,"n_constructed":11,"n_ran_checked":12,"n_instrument":1,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":6,"phrase":"13 ran (of which 11 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/uncertainty-based-offline-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2110.01548","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.01548"}},"official":null}},{"url":"/paper/learning-to-walk-in-minutes-using-massively","slug":"learning-to-walk-in-minutes-using-massively","title":"Learning to Walk in Minutes Using Massively Parallel Deep Reinforcement Learning","date":"2021-09-24","arxiv_id":"2109.11978","repositories_listed":5,"syntology":null},{"url":"/paper/safe-learning-in-robotics-from-learning-based","slug":"safe-learning-in-robotics-from-learning-based","title":"Safe Learning in Robotics: From Learning-Based Control to Safe Reinforcement Learning","date":"2021-08-13","arxiv_id":"2108.06266","repositories_listed":5,"syntology":null},{"url":"/paper/towards-mental-time-travel-a-hierarchical","slug":"towards-mental-time-travel-a-hierarchical","title":"Towards mental time travel: a hierarchical memory for reinforcement learning agents","date":"2021-05-28","arxiv_id":"2105.14039","repositories_listed":5,"syntology":{"n":6,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/towards-mental-time-travel-a-hierarchical#ran","syntology_url":"https://syntology.ai/paper/2105.14039","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.14039"}},"official":{"repos":["deepmind/deepmind-research","deepmind/dm_fast_mapping"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/gym-m-rts-toward-affordable-full-game-real","slug":"gym-m-rts-toward-affordable-full-game-real","title":"Gym-$μ$RTS: Toward Affordable Full Game Real-time Strategy Games Research with Deep Reinforcement Learning","date":"2021-05-21","arxiv_id":"2105.13807","repositories_listed":5,"syntology":null},{"url":"/paper/evolving-reinforcement-learning-algorithms-1","slug":"evolving-reinforcement-learning-algorithms-1","title":"Evolving Reinforcement Learning Algorithms","date":"2021-01-08","arxiv_id":"2101.03958","repositories_listed":5,"syntology":null},{"url":"/paper/smarts-scalable-multi-agent-reinforcement","slug":"smarts-scalable-multi-agent-reinforcement","title":"SMARTS: Scalable Multi-Agent Reinforcement Learning Training School for Autonomous Driving","date":"2020-10-19","arxiv_id":"2010.09776","repositories_listed":5,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/smarts-scalable-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2010.09776","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09776"}},"official":{"repos":["huawei-noah/SMARTS"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/critic-regularized-regression","slug":"critic-regularized-regression","title":"Critic Regularized Regression","date":"2020-06-26","arxiv_id":"2006.15134","repositories_listed":5,"syntology":null},{"url":"/paper/learning-with-amigo-adversarially-motivated","slug":"learning-with-amigo-adversarially-motivated","title":"Learning with AMIGo: Adversarially Motivated Intrinsic Goals","date":"2020-06-22","arxiv_id":"2006.12122","repositories_listed":5,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-with-amigo-adversarially-motivated#ran","syntology_url":"https://syntology.ai/paper/2006.12122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.12122"}},"official":{"repos":["facebookresearch/adversarially-motivated-intrinsic-goals"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/acme-a-research-framework-for-distributed","slug":"acme-a-research-framework-for-distributed","title":"Acme: A Research Framework for Distributed Reinforcement Learning","date":"2020-06-01","arxiv_id":"2006.00979","repositories_listed":5,"syntology":null},{"url":"/paper/agent57-outperforming-the-atari-human","slug":"agent57-outperforming-the-atari-human","title":"Agent57: Outperforming the Atari Human Benchmark","date":"2020-03-30","arxiv_id":"2003.13350","repositories_listed":5,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/agent57-outperforming-the-atari-human#ran","syntology_url":"https://syntology.ai/paper/2003.13350","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.13350"}},"official":null}},{"url":"/paper/stabilizing-transformers-for-reinforcement-1","slug":"stabilizing-transformers-for-reinforcement-1","title":"Stabilizing Transformers for Reinforcement Learning","date":"2019-10-13","arxiv_id":"1910.06764","repositories_listed":5,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stabilizing-transformers-for-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/1910.06764","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.06764"}},"official":null}},{"url":"/paper/a-simple-randomization-technique-for-1","slug":"a-simple-randomization-technique-for-1","title":"Network Randomization: A Simple Technique for Generalization in Deep Reinforcement Learning","date":"2019-10-11","arxiv_id":"1910.05396","repositories_listed":5,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-simple-randomization-technique-for-1#ran","syntology_url":"https://syntology.ai/paper/1910.05396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.05396"}},"official":{"repos":["pokaxpoka/netrand"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-batch-deep-reinforcement","slug":"benchmarking-batch-deep-reinforcement","title":"Benchmarking Batch Deep Reinforcement Learning Algorithms","date":"2019-10-03","arxiv_id":"1910.01708","repositories_listed":5,"syntology":null},{"url":"/paper/multi-agent-deep-reinforcement-learning-for-3","slug":"multi-agent-deep-reinforcement-learning-for-3","title":"Multi-Agent Deep Reinforcement Learning for Liquidation Strategy Analysis","date":"2019-06-24","arxiv_id":"1906.11046","repositories_listed":5,"syntology":null},{"url":"/paper/sqil-imitation-learning-via-regularized","slug":"sqil-imitation-learning-via-regularized","title":"SQIL: Imitation Learning via Reinforcement Learning with Sparse Rewards","date":"2019-05-27","arxiv_id":"1905.11108","repositories_listed":5,"syntology":null},{"url":"/paper/crossnorm-normalization-for-off-policy-td","slug":"crossnorm-normalization-for-off-policy-td","title":"CrossQ: Batch Normalization in Deep Reinforcement Learning for Greater Sample Efficiency and Simplicity","date":"2019-02-14","arxiv_id":"1902.05605","repositories_listed":5,"syntology":null},{"url":"/paper/decoupling-feature-extraction-from-policy","slug":"decoupling-feature-extraction-from-policy","title":"Decoupling feature extraction from policy learning: assessing benefits of state representation learning in goal based robotics","date":"2019-01-24","arxiv_id":"1901.08651","repositories_listed":5,"syntology":null},{"url":"/paper/an-introduction-to-deep-reinforcement","slug":"an-introduction-to-deep-reinforcement","title":"An Introduction to Deep Reinforcement Learning","date":"2018-11-30","arxiv_id":"1811.12560","repositories_listed":5,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based","slug":"deep-reinforcement-learning-based","title":"Deep Reinforcement Learning based Recommendation with Explicit User-Item Interactions Modeling","date":"2018-10-29","arxiv_id":"1810.12027","repositories_listed":5,"syntology":null},{"url":"/paper/deep-reinforcement-learning","slug":"deep-reinforcement-learning","title":"Deep Reinforcement Learning","date":"2018-10-15","arxiv_id":"1810.06339","repositories_listed":5,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1810.06339","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.06339"}},"official":{"repos":["deepmind/lab","google/dopamine","mrkulk/hierarchical-deep-RL","rllab/rllab"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/parametrized-deep-q-networks-learning","slug":"parametrized-deep-q-networks-learning","title":"Parametrized Deep Q-Networks Learning: Reinforcement Learning with Discrete-Continuous Hybrid Action Space","date":"2018-10-10","arxiv_id":"1810.06394","repositories_listed":5,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/parametrized-deep-q-networks-learning#ran","syntology_url":"https://syntology.ai/paper/1810.06394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.06394"}},"official":null}},{"url":"/paper/s-rl-toolbox-environments-datasets-and","slug":"s-rl-toolbox-environments-datasets-and","title":"S-RL Toolbox: Environments, Datasets and Evaluation Metrics for State Representation Learning","date":"2018-09-25","arxiv_id":"1809.09369","repositories_listed":5,"syntology":null},{"url":"/paper/adversarial-deep-reinforcement-learning-in","slug":"adversarial-deep-reinforcement-learning-in","title":"Adversarial Deep Reinforcement Learning in Portfolio Management","date":"2018-08-29","arxiv_id":"1808.09940","repositories_listed":5,"syntology":null},{"url":"/paper/distributed-distributional-deterministic","slug":"distributed-distributional-deterministic","title":"Distributed Distributional Deterministic Policy Gradients","date":"2018-04-23","arxiv_id":"1804.08617","repositories_listed":5,"syntology":null},{"url":"/paper/reinforcement-learning-on-web-interfaces","slug":"reinforcement-learning-on-web-interfaces","title":"Reinforcement Learning on Web Interfaces Using Workflow-Guided Exploration","date":"2018-02-24","arxiv_id":"1802.08802","repositories_listed":5,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reinforcement-learning-on-web-interfaces#ran","syntology_url":"https://syntology.ai/paper/1802.08802","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.08802"}},"official":{"repos":["stanfordnlp/wge","worksheets.codalab.org/worksheets/0x0f25031bd42f4aabbc17625fe1484066"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/fully-decentralized-multi-agent-reinforcement","slug":"fully-decentralized-multi-agent-reinforcement","title":"Fully Decentralized Multi-Agent Reinforcement Learning with Networked Agents","date":"2018-02-23","arxiv_id":"1802.08757","repositories_listed":5,"syntology":null},{"url":"/paper/action-branching-architectures-for-deep","slug":"action-branching-architectures-for-deep","title":"Action Branching Architectures for Deep Reinforcement Learning","date":"2017-11-24","arxiv_id":"1711.08946","repositories_listed":5,"syntology":null},{"url":"/paper/deep-learning-based-numerical-methods-for","slug":"deep-learning-based-numerical-methods-for","title":"Deep learning-based numerical methods for high-dimensional parabolic partial differential equations and backward stochastic differential equations","date":"2017-06-15","arxiv_id":"1706.04702","repositories_listed":5,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/deep-learning-based-numerical-methods-for#ran","syntology_url":"https://syntology.ai/paper/1706.04702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1706.04702"}},"official":null}},{"url":"/paper/concrete-dropout","slug":"concrete-dropout","title":"Concrete Dropout","date":"2017-05-22","arxiv_id":"1705.07832","repositories_listed":5,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 3 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/concrete-dropout#ran","syntology_url":"https://syntology.ai/paper/1705.07832","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1705.07832"}},"official":null}},{"url":"/paper/efficient-parallel-methods-for-deep","slug":"efficient-parallel-methods-for-deep","title":"Efficient Parallel Methods for Deep Reinforcement Learning","date":"2017-05-13","arxiv_id":"1705.04862","repositories_listed":5,"syntology":null},{"url":"/paper/stabilising-experience-replay-for-deep-multi","slug":"stabilising-experience-replay-for-deep-multi","title":"Stabilising Experience Replay for Deep Multi-Agent Reinforcement Learning","date":"2017-02-28","arxiv_id":"1702.08887","repositories_listed":5,"syntology":null},{"url":"/paper/designing-neural-network-architectures-using","slug":"designing-neural-network-architectures-using","title":"Designing Neural Network Architectures using Reinforcement Learning","date":"2016-11-07","arxiv_id":"1611.02167","repositories_listed":5,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/designing-neural-network-architectures-using#ran","syntology_url":"https://syntology.ai/paper/1611.02167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1611.02167"}},"official":null}},{"url":"/paper/multiple-object-recognition-with-visual","slug":"multiple-object-recognition-with-visual","title":"Multiple Object Recognition with Visual Attention","date":"2014-12-24","arxiv_id":"1412.7755","repositories_listed":5,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multiple-object-recognition-with-visual#ran","syntology_url":"https://syntology.ai/paper/1412.7755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1412.7755"}},"official":null}},{"url":"/paper/hierarchical-reinforcement-learning-with-the","slug":"hierarchical-reinforcement-learning-with-the","title":"Hierarchical Reinforcement Learning with the MAXQ Value Function Decomposition","date":"1999-05-21","arxiv_id":"cs/9905014","repositories_listed":5,"syntology":null},{"url":"/paper/logic-rl-unleashing-llm-reasoning-with-rule","slug":"logic-rl-unleashing-llm-reasoning-with-rule","title":"Logic-RL: Unleashing LLM Reasoning with Rule-Based Reinforcement Learning","date":"2025-02-20","arxiv_id":"2502.14768","repositories_listed":4,"syntology":{"n":23,"n_ran":15,"n_constructed":0,"n_ran_checked":13,"n_instrument":2,"n_unverified":8,"n_honours":1,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 0 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/logic-rl-unleashing-llm-reasoning-with-rule#ran","syntology_url":"https://syntology.ai/paper/2502.14768","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14768"}},"official":{"repos":["Unakar/Logic-RL"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":6,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/deepseek-r1-incentivizing-reasoning","slug":"deepseek-r1-incentivizing-reasoning","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","date":"2025-01-22","arxiv_id":"2501.12948","repositories_listed":4,"syntology":null},{"url":"/paper/sigmarl-a-sample-efficient-and-generalizable","slug":"sigmarl-a-sample-efficient-and-generalizable","title":"SigmaRL: A Sample-Efficient and Generalizable Multi-Agent Reinforcement Learning Framework for Motion Planning","date":"2024-08-14","arxiv_id":"2408.07644","repositories_listed":4,"syntology":null},{"url":"/paper/offline-rl-with-no-ood-actions-in-sample","slug":"offline-rl-with-no-ood-actions-in-sample","title":"Offline RL with No OOD Actions: In-Sample Learning via Implicit Value Regularization","date":"2023-03-28","arxiv_id":"2303.15810","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/offline-rl-with-no-ood-actions-in-sample#ran","syntology_url":"https://syntology.ai/paper/2303.15810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.15810"}},"official":{"repos":["ryanxhr/ivr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-in-sample-softmax-for-offline","slug":"the-in-sample-softmax-for-offline","title":"The In-Sample Softmax for Offline Reinforcement Learning","date":"2023-02-28","arxiv_id":"2302.14372","repositories_listed":4,"syntology":null},{"url":"/paper/extreme-q-learning-maxent-rl-without-entropy","slug":"extreme-q-learning-maxent-rl-without-entropy","title":"Extreme Q-Learning: MaxEnt RL without Entropy","date":"2023-01-05","arxiv_id":"2301.02328","repositories_listed":4,"syntology":{"n":13,"n_ran":8,"n_constructed":5,"n_ran_checked":5,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"8 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/extreme-q-learning-maxent-rl-without-entropy#ran","syntology_url":"https://syntology.ai/paper/2301.02328","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.02328"}},"official":null}},{"url":"/paper/lifelong-reinforcement-learning-with","slug":"lifelong-reinforcement-learning-with","title":"Lifelong Reinforcement Learning with Modulating Masks","date":"2022-12-21","arxiv_id":"2212.11110","repositories_listed":4,"syntology":null},{"url":"/paper/finrl-meta-market-environments-and-benchmarks","slug":"finrl-meta-market-environments-and-benchmarks","title":"FinRL-Meta: Market Environments and Benchmarks for Data-Driven Financial Reinforcement Learning","date":"2022-11-06","arxiv_id":"2211.03107","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/finrl-meta-market-environments-and-benchmarks#ran","syntology_url":"https://syntology.ai/paper/2211.03107","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.03107"}},"official":{"repos":["AI4Finance-Foundation/FinRL","ai4finance-foundation/finrl-meta"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/training-a-helpful-and-harmless-assistant","slug":"training-a-helpful-and-harmless-assistant","title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","date":"2022-04-12","arxiv_id":"2204.05862","repositories_listed":4,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/training-a-helpful-and-harmless-assistant#ran","syntology_url":"https://syntology.ai/paper/2204.05862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.05862"}},"official":{"repos":["anthropics/hh-rlhf"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/decom-decomposed-policy-for-constrained","slug":"decom-decomposed-policy-for-constrained","title":"DeCOM: Decomposed Policy for Constrained Cooperative Multi-Agent Reinforcement Learning","date":"2021-11-10","arxiv_id":"2111.05670","repositories_listed":4,"syntology":null},{"url":"/paper/multi-agent-constrained-policy-optimisation","slug":"multi-agent-constrained-policy-optimisation","title":"Multi-Agent Constrained Policy Optimisation","date":"2021-10-06","arxiv_id":"2110.02793","repositories_listed":4,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/multi-agent-constrained-policy-optimisation#ran","syntology_url":"https://syntology.ai/paper/2110.02793","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.02793"}},"official":{"repos":["chauncygu/multi-agent-constrained-policy-optimisation"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/replay-guided-adversarial-environment-design","slug":"replay-guided-adversarial-environment-design","title":"Replay-Guided Adversarial Environment Design","date":"2021-10-06","arxiv_id":"2110.02439","repositories_listed":4,"syntology":{"n":4,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/replay-guided-adversarial-environment-design#ran","syntology_url":"https://syntology.ai/paper/2110.02439","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.02439"}},"official":null}},{"url":"/paper/safe-control-gym-a-unified-benchmark-suite","slug":"safe-control-gym-a-unified-benchmark-suite","title":"safe-control-gym: a Unified Benchmark Suite for Safe Learning-based Control and Reinforcement Learning in Robotics","date":"2021-09-13","arxiv_id":"2109.06325","repositories_listed":4,"syntology":{"n":20,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/safe-control-gym-a-unified-benchmark-suite#ran","syntology_url":"https://syntology.ai/paper/2109.06325","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.06325"}},"official":{"repos":["utiasDSL/safe-control-gym"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/efficient-active-search-for-combinatorial","slug":"efficient-active-search-for-combinatorial","title":"Efficient Active Search for Combinatorial Optimization Problems","date":"2021-06-09","arxiv_id":"2106.05126","repositories_listed":4,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/efficient-active-search-for-combinatorial#ran","syntology_url":"https://syntology.ai/paper/2106.05126","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.05126"}},"official":{"repos":["ahottung/EAS"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/objective-robustness-in-deep-reinforcement","slug":"objective-robustness-in-deep-reinforcement","title":"Goal Misgeneralization in Deep Reinforcement Learning","date":"2021-05-28","arxiv_id":"2105.14111","repositories_listed":4,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/objective-robustness-in-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2105.14111","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.14111"}},"official":{"repos":["JacobPfau/procgenAISC","jbkjr/train-procgen-pytorch"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a-reinforcement-learning-environment-for-job","slug":"a-reinforcement-learning-environment-for-job","title":"A Reinforcement Learning Environment For Job-Shop Scheduling","date":"2021-04-08","arxiv_id":"2104.03760","repositories_listed":4,"syntology":null},{"url":"/paper/amp-adversarial-motion-priors-for-stylized","slug":"amp-adversarial-motion-priors-for-stylized","title":"AMP: Adversarial Motion Priors for Stylized Physics-Based Character Control","date":"2021-04-05","arxiv_id":"2104.02180","repositories_listed":4,"syntology":null},{"url":"/paper/recurrent-rational-networks","slug":"recurrent-rational-networks","title":"Adaptive Rational Activations to Boost Deep Reinforcement Learning","date":"2021-02-18","arxiv_id":"2102.09407","repositories_listed":4,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recurrent-rational-networks#ran","syntology_url":"https://syntology.ai/paper/2102.09407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.09407"}},"official":{"repos":["ml-research/rational_activations","ml-research/rational_rl","ml-research/rational_sl","k4ntz/activation-functions"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-distracting-control-suite-a-challenging","slug":"the-distracting-control-suite-a-challenging","title":"The Distracting Control Suite -- A Challenging Benchmark for Reinforcement Learning from Pixels","date":"2021-01-07","arxiv_id":"2101.02722","repositories_listed":4,"syntology":null},{"url":"/paper/learning-to-dispatch-for-job-shop-scheduling","slug":"learning-to-dispatch-for-job-shop-scheduling","title":"Learning to Dispatch for Job Shop Scheduling via Deep Reinforcement Learning","date":"2020-10-23","arxiv_id":"2010.12367","repositories_listed":4,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-to-dispatch-for-job-shop-scheduling#ran","syntology_url":"https://syntology.ai/paper/2010.12367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.12367"}},"official":{"repos":["zcajiayin/L2D"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/d2rl-deep-dense-architectures-in-1","slug":"d2rl-deep-dense-architectures-in-1","title":"D2RL: Deep Dense Architectures in Reinforcement Learning","date":"2020-10-19","arxiv_id":"2010.09163","repositories_listed":4,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/d2rl-deep-dense-architectures-in-1#ran","syntology_url":"https://syntology.ai/paper/2010.09163","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09163"}},"official":{"repos":["pairlab/d2rl"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["found_in_text","listed","official"]}}}],"record_sha256":"7f3e18e9a87e7cf45b1c24e175d65354eea3d0cdb4151c48e077fdfa6a64f210","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}