{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/4","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":135,"rows_per_page":100,"rows":[301,400],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/3","next":"/task/reinforcement-learning-2/papers/5","papers":[{"url":"/paper/quota-the-quantile-option-architecture-for","slug":"quota-the-quantile-option-architecture-for","title":"QUOTA: The Quantile Option Architecture for Reinforcement Learning","date":"2018-11-05","arxiv_id":"1811.02073","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/quota-the-quantile-option-architecture-for#ran","syntology_url":"https://syntology.ai/paper/1811.02073","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.02073"}},"official":{"repos":["ShangtongZhang/DeepRL"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/social-influence-as-intrinsic-motivation-for","slug":"social-influence-as-intrinsic-motivation-for","title":"Social Influence as Intrinsic Motivation for Multi-Agent Deep Reinforcement Learning","date":"2018-10-19","arxiv_id":"1810.08647","repositories_listed":3,"syntology":null},{"url":"/paper/actor-attention-critic-for-multi-agent","slug":"actor-attention-critic-for-multi-agent","title":"Actor-Attention-Critic for Multi-Agent Reinforcement Learning","date":"2018-10-05","arxiv_id":"1810.02912","repositories_listed":3,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/actor-attention-critic-for-multi-agent#ran","syntology_url":"https://syntology.ai/paper/1810.02912","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.02912"}},"official":{"repos":["shariqiqbal2810/MAAC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/deep-quality-value-dqv-learning","slug":"deep-quality-value-dqv-learning","title":"Deep Quality-Value (DQV) Learning","date":"2018-09-30","arxiv_id":"1810.00368","repositories_listed":3,"syntology":null},{"url":"/paper/dynamic-weights-in-multi-objective-deep","slug":"dynamic-weights-in-multi-objective-deep","title":"Dynamic Weights in Multi-Objective Deep Reinforcement Learning","date":"2018-09-20","arxiv_id":"1809.07803","repositories_listed":3,"syntology":null},{"url":"/paper/decoupling-strategy-and-generation-in","slug":"decoupling-strategy-and-generation-in","title":"Decoupling Strategy and Generation in Negotiation Dialogues","date":"2018-08-29","arxiv_id":"1808.09637","repositories_listed":3,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/decoupling-strategy-and-generation-in#ran","syntology_url":"https://syntology.ai/paper/1808.09637","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.09637"}},"official":{"repos":["worksheets.codalab.org/worksheets/0x453913e76b65495d8b9730d41c7e0a0c"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/bipedal-walking-robot-using-deep","slug":"bipedal-walking-robot-using-deep","title":"Bipedal Walking Robot using Deep Deterministic Policy Gradient","date":"2018-07-16","arxiv_id":"1807.05924","repositories_listed":3,"syntology":null},{"url":"/paper/scheduled-policy-optimization-for-natural","slug":"scheduled-policy-optimization-for-natural","title":"Scheduled Policy Optimization for Natural Language Communication with Intelligent Agents","date":"2018-06-16","arxiv_id":"1806.06187","repositories_listed":3,"syntology":null},{"url":"/paper/maximum-a-posteriori-policy-optimisation","slug":"maximum-a-posteriori-policy-optimisation","title":"Maximum a Posteriori Policy Optimisation","date":"2018-06-14","arxiv_id":"1806.06920","repositories_listed":3,"syntology":null},{"url":"/paper/transfer-learning-for-related-reinforcement","slug":"transfer-learning-for-related-reinforcement","title":"Transfer Learning for Related Reinforcement Learning Tasks via Image-to-Image Translation","date":"2018-05-31","arxiv_id":"1806.07377","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-sequence-to","slug":"deep-reinforcement-learning-for-sequence-to","title":"Deep Reinforcement Learning For Sequence to Sequence Models","date":"2018-05-24","arxiv_id":"1805.09461","repositories_listed":3,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":15,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":2,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deep-reinforcement-learning-for-sequence-to#ran","syntology_url":"https://syntology.ai/paper/1805.09461","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.09461"}},"official":{"repos":["yaserkl/RLSeq2Seq"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/toward-diverse-text-generation-with-inverse","slug":"toward-diverse-text-generation-with-inverse","title":"Toward Diverse Text Generation with Inverse Reinforcement Learning","date":"2018-04-30","arxiv_id":"1804.11258","repositories_listed":3,"syntology":null},{"url":"/paper/gotta-learn-fast-a-new-benchmark-for","slug":"gotta-learn-fast-a-new-benchmark-for","title":"Gotta Learn Fast: A New Benchmark for Generalization in RL","date":"2018-04-10","arxiv_id":"1804.03720","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-traffic-light","slug":"deep-reinforcement-learning-for-traffic-light","title":"Deep Reinforcement Learning for Traffic Light Control in Vehicular Networks","date":"2018-03-29","arxiv_id":"1803.11115","repositories_listed":3,"syntology":null},{"url":"/paper/synthesizing-neural-network-controllers-with","slug":"synthesizing-neural-network-controllers-with","title":"Synthesizing Neural Network Controllers with Probabilistic Model based Reinforcement Learning","date":"2018-03-06","arxiv_id":"1803.02291","repositories_listed":3,"syntology":null},{"url":"/paper/mean-field-multi-agent-reinforcement-learning","slug":"mean-field-multi-agent-reinforcement-learning","title":"Mean Field Multi-Agent Reinforcement Learning","date":"2018-02-15","arxiv_id":"1802.05438","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mean-field-multi-agent-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1802.05438","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.05438"}},"official":{"repos":["mlii/mfrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/jointly-learning-to-construct-and-control","slug":"jointly-learning-to-construct-and-control","title":"Jointly Learning to Construct and Control Agents using Deep Reinforcement Learning","date":"2018-01-04","arxiv_id":"1801.01432","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/jointly-learning-to-construct-and-control#ran","syntology_url":"https://syntology.ai/paper/1801.01432","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1801.01432"}},"official":null}},{"url":"/paper/rllib-abstractions-for-distributed","slug":"rllib-abstractions-for-distributed","title":"RLlib: Abstractions for Distributed Reinforcement Learning","date":"2017-12-26","arxiv_id":"1712.09381","repositories_listed":3,"syntology":null},{"url":"/paper/magent-a-many-agent-reinforcement-learning","slug":"magent-a-many-agent-reinforcement-learning","title":"MAgent: A Many-Agent Reinforcement Learning Platform for Artificial Collective Intelligence","date":"2017-12-02","arxiv_id":"1712.00600","repositories_listed":3,"syntology":null},{"url":"/paper/a2-rl-aesthetics-aware-reinforcement-learning","slug":"a2-rl-aesthetics-aware-reinforcement-learning","title":"A2-RL: Aesthetics Aware Reinforcement Learning for Image Cropping","date":"2017-09-14","arxiv_id":"1709.04595","repositories_listed":3,"syntology":null},{"url":"/paper/chemgan-challenge-for-drug-discovery-can-ai","slug":"chemgan-challenge-for-drug-discovery-can-ai","title":"ChemGAN challenge for drug discovery: can AI reproduce natural chemical diversity?","date":"2017-08-28","arxiv_id":"1708.08227","repositories_listed":3,"syntology":null},{"url":"/paper/efficient-architecture-search-by-network","slug":"efficient-architecture-search-by-network","title":"Efficient Architecture Search by Network Transformation","date":"2017-07-16","arxiv_id":"1707.04873","repositories_listed":3,"syntology":null},{"url":"/paper/efficient-probabilistic-performance-bounds","slug":"efficient-probabilistic-performance-bounds","title":"Efficient Probabilistic Performance Bounds for Inverse Reinforcement Learning","date":"2017-07-03","arxiv_id":"1707.00724","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-probabilistic-performance-bounds#ran","syntology_url":"https://syntology.ai/paper/1707.00724","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1707.00724"}},"official":{"repos":["dsbrown1331/aaai-2018-code"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/reinforced-mnemonic-reader-for-machine","slug":"reinforced-mnemonic-reader-for-machine","title":"Reinforced Mnemonic Reader for Machine Reading Comprehension","date":"2017-05-08","arxiv_id":"1705.02798","repositories_listed":3,"syntology":null},{"url":"/paper/from-language-to-programs-bridging","slug":"from-language-to-programs-bridging","title":"From Language to Programs: Bridging Reinforcement Learning and Maximum Marginal Likelihood","date":"2017-04-25","arxiv_id":"1704.07926","repositories_listed":3,"syntology":null},{"url":"/paper/molecular-de-novo-design-through-deep","slug":"molecular-de-novo-design-through-deep","title":"Molecular De Novo Design through Deep Reinforcement Learning","date":"2017-04-25","arxiv_id":"1704.07555","repositories_listed":3,"syntology":null},{"url":"/paper/hybrid-code-networks-practical-and-efficient","slug":"hybrid-code-networks-practical-and-efficient","title":"Hybrid Code Networks: practical and efficient end-to-end dialog control with supervised and reinforcement learning","date":"2017-02-10","arxiv_id":"1702.03274","repositories_listed":3,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hybrid-code-networks-practical-and-efficient#ran","syntology_url":"https://syntology.ai/paper/1702.03274","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1702.03274"}},"official":null}},{"url":"/paper/an-alternative-softmax-operator-for","slug":"an-alternative-softmax-operator-for","title":"An Alternative Softmax Operator for Reinforcement Learning","date":"2016-12-16","arxiv_id":"1612.05628","repositories_listed":3,"syntology":null},{"url":"/paper/cryptocurrency-portfolio-management-with-deep","slug":"cryptocurrency-portfolio-management-with-deep","title":"Cryptocurrency Portfolio Management with Deep Reinforcement Learning","date":"2016-12-05","arxiv_id":"1612.01277","repositories_listed":3,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cryptocurrency-portfolio-management-with-deep#ran","syntology_url":"https://syntology.ai/paper/1612.01277","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1612.01277"}},"official":{"repos":["ZhengyaoJiang/PGPortfolio"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/reinforcement-learning-through-asynchronous","slug":"reinforcement-learning-through-asynchronous","title":"Reinforcement Learning through Asynchronous Advantage Actor-Critic on a GPU","date":"2016-11-18","arxiv_id":"1611.06256","repositories_listed":3,"syntology":null},{"url":"/paper/reinforcement-learning-with-unsupervised","slug":"reinforcement-learning-with-unsupervised","title":"Reinforcement Learning with Unsupervised Auxiliary Tasks","date":"2016-11-16","arxiv_id":"1611.05397","repositories_listed":3,"syntology":null},{"url":"/paper/exploration-a-study-of-count-based","slug":"exploration-a-study-of-count-based","title":"#Exploration: A Study of Count-Based Exploration for Deep Reinforcement Learning","date":"2016-11-15","arxiv_id":"1611.04717","repositories_listed":3,"syntology":null},{"url":"/paper/a-connection-between-generative-adversarial","slug":"a-connection-between-generative-adversarial","title":"A Connection between Generative Adversarial Networks, Inverse Reinforcement Learning, and Energy-Based Models","date":"2016-11-11","arxiv_id":"1611.03852","repositories_listed":3,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/a-connection-between-generative-adversarial#ran","syntology_url":"https://syntology.ai/paper/1611.03852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1611.03852"}},"official":null}},{"url":"/paper/input-convex-neural-networks","slug":"input-convex-neural-networks","title":"Input Convex Neural Networks","date":"2016-09-22","arxiv_id":"1609.07152","repositories_listed":3,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/input-convex-neural-networks#ran","syntology_url":"https://syntology.ai/paper/1609.07152","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1609.07152"}},"official":{"repos":["locuslab/icnn"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-free-episodic-control","slug":"model-free-episodic-control","title":"Model-Free Episodic Control","date":"2016-06-14","arxiv_id":"1606.04460","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/model-free-episodic-control#ran","syntology_url":"https://syntology.ai/paper/1606.04460","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1606.04460"}},"official":null}},{"url":"/paper/safe-and-efficient-off-policy-reinforcement","slug":"safe-and-efficient-off-policy-reinforcement","title":"Safe and Efficient Off-Policy Reinforcement Learning","date":"2016-06-08","arxiv_id":"1606.02647","repositories_listed":3,"syntology":null},{"url":"/paper/learning-to-communicate-with-deep-multi-agent","slug":"learning-to-communicate-with-deep-multi-agent","title":"Learning to Communicate with Deep Multi-Agent Reinforcement Learning","date":"2016-05-21","arxiv_id":"1605.06676","repositories_listed":3,"syntology":null},{"url":"/paper/data-efficient-off-policy-policy-evaluation","slug":"data-efficient-off-policy-policy-evaluation","title":"Data-Efficient Off-Policy Policy Evaluation for Reinforcement Learning","date":"2016-04-04","arxiv_id":"1604.00923","repositories_listed":3,"syntology":null},{"url":"/paper/learning-to-compose-neural-networks-for","slug":"learning-to-compose-neural-networks-for","title":"Learning to Compose Neural Networks for Question Answering","date":"2016-01-07","arxiv_id":"1601.01705","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-compose-neural-networks-for#ran","syntology_url":"https://syntology.ai/paper/1601.01705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1601.01705"}},"official":{"repos":["jacobandreas/nmn2"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/taming-the-noise-in-reinforcement-learning","slug":"taming-the-noise-in-reinforcement-learning","title":"Taming the Noise in Reinforcement Learning via Soft Updates","date":"2015-12-28","arxiv_id":"1512.08562","repositories_listed":3,"syntology":null},{"url":"/paper/deep-attention-recurrent-q-network","slug":"deep-attention-recurrent-q-network","title":"Deep Attention Recurrent Q-Network","date":"2015-12-05","arxiv_id":"1512.01693","repositories_listed":3,"syntology":null},{"url":"/paper/actor-mimic-deep-multitask-and-transfer","slug":"actor-mimic-deep-multitask-and-transfer","title":"Actor-Mimic: Deep Multitask and Transfer Reinforcement Learning","date":"2015-11-19","arxiv_id":"1511.06342","repositories_listed":3,"syntology":null},{"url":"/paper/active-object-localization-with-deep","slug":"active-object-localization-with-deep","title":"Active Object Localization with Deep Reinforcement Learning","date":"2015-11-18","arxiv_id":"1511.06015","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-a-natural","slug":"deep-reinforcement-learning-with-a-natural","title":"Deep Reinforcement Learning with a Natural Language Action Space","date":"2015-11-14","arxiv_id":"1511.04636","repositories_listed":3,"syntology":null},{"url":"/paper/reinforcement-learning-with-parameterized","slug":"reinforcement-learning-with-parameterized","title":"Reinforcement Learning with Parameterized Actions","date":"2015-09-05","arxiv_id":"1509.01644","repositories_listed":3,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-parameterized#ran","syntology_url":"https://syntology.ai/paper/1509.01644","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1509.01644"}},"official":null}},{"url":"/paper/giraffe-using-deep-reinforcement-learning-to","slug":"giraffe-using-deep-reinforcement-learning-to","title":"Giraffe: Using Deep Reinforcement Learning to Play Chess","date":"2015-09-04","arxiv_id":"1509.01549","repositories_listed":3,"syntology":null},{"url":"/paper/massively-parallel-methods-for-deep","slug":"massively-parallel-methods-for-deep","title":"Massively Parallel Methods for Deep Reinforcement Learning","date":"2015-07-15","arxiv_id":"1507.04296","repositories_listed":3,"syntology":null},{"url":"/paper/language-understanding-for-text-based-games","slug":"language-understanding-for-text-based-games","title":"Language Understanding for Text-based Games Using Deep Reinforcement Learning","date":"2015-06-30","arxiv_id":"1506.08941","repositories_listed":3,"syntology":null},{"url":"/paper/the-arcade-learning-environment-an-evaluation","slug":"the-arcade-learning-environment-an-evaluation","title":"The Arcade Learning Environment: An Evaluation Platform for General Agents","date":"2012-07-19","arxiv_id":"1207.4708","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-arcade-learning-environment-an-evaluation#ran","syntology_url":"https://syntology.ai/paper/1207.4708","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1207.4708"}},"official":null}},{"url":"/paper/co-evolving-llm-coder-and-unit-tester-via","slug":"co-evolving-llm-coder-and-unit-tester-via","title":"Co-Evolving LLM Coder and Unit Tester via Reinforcement Learning","date":"2025-06-03","arxiv_id":"2506.03136","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/co-evolving-llm-coder-and-unit-tester-via#ran","syntology_url":"https://syntology.ai/paper/2506.03136","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.03136"}},"official":{"repos":["gen-verse/cure"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/advancing-multimodal-reasoning-via","slug":"advancing-multimodal-reasoning-via","title":"Advancing Multimodal Reasoning via Reinforcement Learning with Cold Start","date":"2025-05-28","arxiv_id":"2505.22334","repositories_listed":2,"syntology":null},{"url":"/paper/4hammer-a-board-game-reinforcement-learning","slug":"4hammer-a-board-game-reinforcement-learning","title":"4Hammer: a board-game reinforcement learning environment for the hour long time frame","date":"2025-05-19","arxiv_id":"2505.13638","repositories_listed":2,"syntology":null},{"url":"/paper/2505-11289","slug":"2505-11289","title":"Meta-World+: An Improved, Standardized, RL Benchmark","date":"2025-05-16","arxiv_id":"2505.11289","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2505-11289#ran","syntology_url":"https://syntology.ai/paper/2505.11289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.11289"}},"official":{"repos":["farama-foundation/metaworld"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-clean-slate-for-offline-reinforcement","slug":"a-clean-slate-for-offline-reinforcement","title":"A Clean Slate for Offline Reinforcement Learning","date":"2025-04-15","arxiv_id":"2504.11453","repositories_listed":2,"syntology":null},{"url":"/paper/lightweight-and-direct-document-relevance","slug":"lightweight-and-direct-document-relevance","title":"Lightweight and Direct Document Relevance Optimization for Generative Information Retrieval","date":"2025-04-07","arxiv_id":"2504.05181","repositories_listed":2,"syntology":null},{"url":"/paper/towards-optimal-adversarial-robust","slug":"towards-optimal-adversarial-robust","title":"Towards Optimal Adversarial Robust Reinforcement Learning with Infinity Measurement Error","date":"2025-02-23","arxiv_id":"2502.16734","repositories_listed":2,"syntology":null},{"url":"/paper/evorl-a-gpu-accelerated-framework-for","slug":"evorl-a-gpu-accelerated-framework-for","title":"EvoRL: A GPU-accelerated Framework for Evolutionary Reinforcement Learning","date":"2025-01-25","arxiv_id":"2501.15129","repositories_listed":2,"syntology":null},{"url":"/paper/offline-reinforcement-learning-for-llm-multi","slug":"offline-reinforcement-learning-for-llm-multi","title":"Offline Reinforcement Learning for LLM Multi-Step Reasoning","date":"2024-12-20","arxiv_id":"2412.16145","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/offline-reinforcement-learning-for-llm-multi#ran","syntology_url":"https://syntology.ai/paper/2412.16145","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.16145"}},"official":{"repos":["jwhj/oreo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/latent-reward-llm-empowered-credit-assignment","slug":"latent-reward-llm-empowered-credit-assignment","title":"Latent Reward: LLM-Empowered Credit Assignment in Episodic Reinforcement Learning","date":"2024-12-15","arxiv_id":"2412.11120","repositories_listed":2,"syntology":null},{"url":"/paper/improve-vision-language-model-chain-of","slug":"improve-vision-language-model-chain-of","title":"Improve Vision Language Model Chain-of-thought Reasoning","date":"2024-10-21","arxiv_id":"2410.16198","repositories_listed":2,"syntology":{"n":22,"n_ran":18,"n_constructed":0,"n_ran_checked":13,"n_instrument":5,"n_unverified":4,"n_honours":0,"n_violates":3,"n_no_contract":10,"n_pointer_only":22,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 3 violated, 10 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/improve-vision-language-model-chain-of#ran","syntology_url":"https://syntology.ai/paper/2410.16198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.16198"}},"official":{"repos":["riflezhang/llava-reasoner-dpo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/it-takes-two-to-tango-directly-optimizing-for","slug":"it-takes-two-to-tango-directly-optimizing-for","title":"It Takes Two to Tango: Directly Optimizing for Constrained Synthesizability in Generative Molecular Design","date":"2024-10-15","arxiv_id":"2410.11527","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/it-takes-two-to-tango-directly-optimizing-for#ran","syntology_url":"https://syntology.ai/paper/2410.11527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11527"}},"official":{"repos":["schwallergroup/saturn"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/enhancing-multi-step-reasoning-abilities-of","slug":"enhancing-multi-step-reasoning-abilities-of","title":"Enhancing Multi-Step Reasoning Abilities of Language Models through Direct Q-Function Optimization","date":"2024-10-11","arxiv_id":"2410.09302","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-based-model-predictive","slug":"reinforcement-learning-based-model-predictive","title":"Reinforcement Learning-based Model Predictive Control for Greenhouse Climate Control","date":"2024-09-19","arxiv_id":"2409.12789","repositories_listed":2,"syntology":null},{"url":"/paper/training-language-models-to-self-correct-via","slug":"training-language-models-to-self-correct-via","title":"Training Language Models to Self-Correct via Reinforcement Learning","date":"2024-09-19","arxiv_id":"2409.12917","repositories_listed":2,"syntology":null},{"url":"/paper/sparsifying-parametric-models-with-l0","slug":"sparsifying-parametric-models-with-l0","title":"Sparsifying Parametric Models with L0 Regularization","date":"2024-09-05","arxiv_id":"2409.03489","repositories_listed":2,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-for-22","slug":"multi-agent-reinforcement-learning-for-22","title":"Multi-Agent Reinforcement Learning for Autonomous Driving: A Survey","date":"2024-08-19","arxiv_id":"2408.09675","repositories_listed":2,"syntology":null},{"url":"/paper/gradient-boosting-reinforcement-learning","slug":"gradient-boosting-reinforcement-learning","title":"Gradient Boosting Reinforcement Learning","date":"2024-07-11","arxiv_id":"2407.08250","repositories_listed":2,"syntology":null},{"url":"/paper/oralytics-reinforcement-learning-algorithm","slug":"oralytics-reinforcement-learning-algorithm","title":"Oralytics Reinforcement Learning Algorithm","date":"2024-06-19","arxiv_id":"2406.13127","repositories_listed":2,"syntology":null},{"url":"/paper/an-imitative-reinforcement-learning-framework","slug":"an-imitative-reinforcement-learning-framework","title":"An Imitative Reinforcement Learning Framework for Autonomous Dogfight","date":"2024-06-17","arxiv_id":"2406.11562","repositories_listed":2,"syntology":null},{"url":"/paper/alignsam-aligning-segment-anything-model-to","slug":"alignsam-aligning-segment-anything-model-to","title":"AlignSAM: Aligning Segment Anything Model to Open Context via Reinforcement Learning","date":"2024-06-01","arxiv_id":"2406.00480","repositories_listed":2,"syntology":null},{"url":"/paper/hope-a-reinforcement-learning-based-hybrid","slug":"hope-a-reinforcement-learning-based-hybrid","title":"HOPE: A Reinforcement Learning-based Hybrid Policy Path Planner for Diverse Parking Scenarios","date":"2024-05-31","arxiv_id":"2405.20579","repositories_listed":2,"syntology":null},{"url":"/paper/q-value-regularized-transformer-for-offline","slug":"q-value-regularized-transformer-for-offline","title":"Q-value Regularized Transformer for Offline Reinforcement Learning","date":"2024-05-27","arxiv_id":"2405.17098","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/q-value-regularized-transformer-for-offline#ran","syntology_url":"https://syntology.ai/paper/2405.17098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17098"}},"official":null}},{"url":"/paper/highway-graph-to-accelerate-reinforcement","slug":"highway-graph-to-accelerate-reinforcement","title":"Highway Graph to Accelerate Reinforcement Learning","date":"2024-05-20","arxiv_id":"2405.11727","repositories_listed":2,"syntology":null},{"url":"/paper/on-robust-reinforcement-learning-with","slug":"on-robust-reinforcement-learning-with","title":"On Robust Reinforcement Learning with Lipschitz-Bounded Policy Networks","date":"2024-05-19","arxiv_id":"2405.11432","repositories_listed":2,"syntology":null},{"url":"/paper/reward-centering","slug":"reward-centering","title":"Reward Centering","date":"2024-05-16","arxiv_id":"2405.09999","repositories_listed":2,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reward-centering#ran","syntology_url":"https://syntology.ai/paper/2405.09999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.09999"}},"official":{"repos":["abhisheknaik96/continuing-rl-exps"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/acegen-reinforcement-learning-of-generative","slug":"acegen-reinforcement-learning-of-generative","title":"ACEGEN: Reinforcement learning of generative chemical agents for drug discovery","date":"2024-05-07","arxiv_id":"2405.04657","repositories_listed":2,"syntology":null},{"url":"/paper/model-based-reinforcement-learning-for-7","slug":"model-based-reinforcement-learning-for-7","title":"Model-based Reinforcement Learning for Parameterized Action Spaces","date":"2024-04-03","arxiv_id":"2404.03037","repositories_listed":2,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-based-reinforcement-learning-for-7#ran","syntology_url":"https://syntology.ai/paper/2404.03037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03037"}},"official":{"repos":["valarzz/dlpa","valarzz/model-based-reinforcement-learning-for-parameterized-action-spaces"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/peersimgym-an-environment-for-solving-the","slug":"peersimgym-an-environment-for-solving-the","title":"PeersimGym: An Environment for Solving the Task Offloading Problem with Reinforcement Learning","date":"2024-03-26","arxiv_id":"2403.17637","repositories_listed":2,"syntology":null},{"url":"/paper/the-edge-of-reach-problem-in-offline-model","slug":"the-edge-of-reach-problem-in-offline-model","title":"The Edge-of-Reach Problem in Offline Model-Based Reinforcement Learning","date":"2024-02-19","arxiv_id":"2402.12527","repositories_listed":2,"syntology":null},{"url":"/paper/performative-reinforcement-learning-in","slug":"performative-reinforcement-learning-in","title":"Performative Reinforcement Learning in Gradually Shifting Environments","date":"2024-02-15","arxiv_id":"2402.09838","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/performative-reinforcement-learning-in#ran","syntology_url":"https://syntology.ai/paper/2402.09838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09838"}},"official":{"repos":["bsen/performative-rl-gradually-shifting-envs","rank-and-files/performative-rl-gradually-shifting-envs"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-shot-reinforcement-learning-via-function","slug":"zero-shot-reinforcement-learning-via-function","title":"Zero-Shot Reinforcement Learning via Function Encoders","date":"2024-01-30","arxiv_id":"2401.17173","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/zero-shot-reinforcement-learning-via-function#ran","syntology_url":"https://syntology.ai/paper/2401.17173","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.17173"}},"official":{"repos":["anonymousresearcher5642/functionencoderrl","tyler-ingebrand/functionencoderrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/off-policy-primal-dual-safe-reinforcement","slug":"off-policy-primal-dual-safe-reinforcement","title":"Off-Policy Primal-Dual Safe Reinforcement Learning","date":"2024-01-26","arxiv_id":"2401.14758","repositories_listed":2,"syntology":{"n":22,"n_ran":13,"n_constructed":1,"n_ran_checked":11,"n_instrument":2,"n_unverified":9,"n_honours":0,"n_violates":1,"n_no_contract":10,"n_pointer_only":10,"phrase":"13 ran (of which 1 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/off-policy-primal-dual-safe-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2401.14758","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.14758"}},"official":{"repos":["pku-alignment/omnisafe","zifanwu/cal"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/autonomous-driving-using-residual-sensor","slug":"autonomous-driving-using-residual-sensor","title":"Autonomous Driving using Residual Sensor Fusion and Deep Reinforcement Learning","date":"2023-12-27","arxiv_id":"2312.16620","repositories_listed":2,"syntology":null},{"url":"/paper/pdit-interleaving-perception-and-decision","slug":"pdit-interleaving-perception-and-decision","title":"PDiT: Interleaving Perception and Decision-making Transformers for Deep Reinforcement Learning","date":"2023-12-26","arxiv_id":"2312.15863","repositories_listed":2,"syntology":null},{"url":"/paper/de-novo-drug-design-using-reinforcement-1","slug":"de-novo-drug-design-using-reinforcement-1","title":"De novo Drug Design using Reinforcement Learning with Multiple GPT Agents","date":"2023-12-21","arxiv_id":"2401.06155","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/de-novo-drug-design-using-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2401.06155","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.06155"}},"official":{"repos":["hxyfighter/molrl-mgpt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/optimizing-heat-alert-issuance-for-public","slug":"optimizing-heat-alert-issuance-for-public","title":"Optimizing Heat Alert Issuance with Reinforcement Learning","date":"2023-12-21","arxiv_id":"2312.14196","repositories_listed":2,"syntology":null},{"url":"/paper/xland-minigrid-scalable-meta-reinforcement","slug":"xland-minigrid-scalable-meta-reinforcement","title":"XLand-MiniGrid: Scalable Meta-Reinforcement Learning Environments in JAX","date":"2023-12-19","arxiv_id":"2312.12044","repositories_listed":2,"syntology":null},{"url":"/paper/active-reinforcement-learning-for-robust","slug":"active-reinforcement-learning-for-robust","title":"Active Reinforcement Learning for Robust Building Control","date":"2023-12-16","arxiv_id":"2312.10289","repositories_listed":2,"syntology":null},{"url":"/paper/rat-reinforcement-learning-driven-and","slug":"rat-reinforcement-learning-driven-and","title":"RAT: Reinforcement-Learning-Driven and Adaptive Testing for Vulnerability Discovery in Web Application Firewalls","date":"2023-12-13","arxiv_id":"2312.07885","repositories_listed":2,"syntology":null},{"url":"/paper/synergizing-quality-diversity-with-descriptor","slug":"synergizing-quality-diversity-with-descriptor","title":"Synergizing Quality-Diversity with Descriptor-Conditioned Reinforcement Learning","date":"2023-12-10","arxiv_id":"2401.08632","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/synergizing-quality-diversity-with-descriptor#ran","syntology_url":"https://syntology.ai/paper/2401.08632","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.08632"}},"official":{"repos":["adaptive-intelligent-robotics/DCRL-MAP-Elites"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/few-shot-multispectral-segmentation-with","slug":"few-shot-multispectral-segmentation-with","title":"Few-shot Multispectral Segmentation with Representations Generated by Reinforcement Learning","date":"2023-11-20","arxiv_id":"2311.11827","repositories_listed":2,"syntology":null},{"url":"/paper/tactics2d-a-multi-agent-reinforcement","slug":"tactics2d-a-multi-agent-reinforcement","title":"Tactics2D: A Highly Modular and Extensible Simulator for Driving Decision-making","date":"2023-11-18","arxiv_id":"2311.11058","repositories_listed":2,"syntology":null},{"url":"/paper/drm-mastering-visual-reinforcement-learning","slug":"drm-mastering-visual-reinforcement-learning","title":"DrM: Mastering Visual Reinforcement Learning through Dormant Ratio Minimization","date":"2023-10-30","arxiv_id":"2310.19668","repositories_listed":2,"syntology":{"n":14,"n_ran":10,"n_constructed":3,"n_ran_checked":8,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":6,"phrase":"10 ran (of which 3 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/drm-mastering-visual-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2310.19668","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19668"}},"official":{"repos":["XuGW-Kevin/DrM"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/dynamics-generalisation-in-reinforcement-1","slug":"dynamics-generalisation-in-reinforcement-1","title":"Dynamics Generalisation in Reinforcement Learning via Adaptive Context-Aware Policies","date":"2023-10-25","arxiv_id":"2310.16686","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":4,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamics-generalisation-in-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2310.16686","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.16686"}},"official":{"repos":["michael-beukman/decisionadapter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/recurrent-linear-transformers","slug":"recurrent-linear-transformers","title":"AGaLiTe: Approximate Gated Linear Transformers for Online Reinforcement Learning","date":"2023-10-24","arxiv_id":"2310.15719","repositories_listed":2,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/recurrent-linear-transformers#ran","syntology_url":"https://syntology.ai/paper/2310.15719","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.15719"}},"official":{"repos":["subho406/Recurrent-Linear-Transformers","subho406/agalite"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-robust-offline-reinforcement-learning","slug":"towards-robust-offline-reinforcement-learning","title":"Towards Robust Offline Reinforcement Learning under Diverse Data Corruption","date":"2023-10-19","arxiv_id":"2310.12955","repositories_listed":2,"syntology":{"n":9,"n_ran":5,"n_constructed":2,"n_ran_checked":3,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":9,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/towards-robust-offline-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2310.12955","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12955"}},"official":{"repos":["yangrui2015/riql"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/rllte-long-term-evolution-project-of","slug":"rllte-long-term-evolution-project-of","title":"RLLTE: Long-Term Evolution Project of Reinforcement Learning","date":"2023-09-28","arxiv_id":"2309.16382","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rllte-long-term-evolution-project-of#ran","syntology_url":"https://syntology.ai/paper/2309.16382","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16382"}},"official":{"repos":["RLE-Foundation/rllte"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-toolkit-for-reliable-benchmarking-and","slug":"a-toolkit-for-reliable-benchmarking-and","title":"A Toolkit for Reliable Benchmarking and Research in Multi-Objective Reinforcement Learning","date":"2023-09-26","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/conservative-world-models","slug":"conservative-world-models","title":"Zero-Shot Reinforcement Learning from Low Quality Data","date":"2023-09-26","arxiv_id":"2309.15178","repositories_listed":2,"syntology":{"n":18,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":8,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/conservative-world-models#ran","syntology_url":"https://syntology.ai/paper/2309.15178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.15178"}},"official":{"repos":["enjeeneer/conservative-world-models","enjeeneer/zero-shot-rl"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-image-to","slug":"deep-reinforcement-learning-for-image-to","title":"RL-I2IT: Image-to-Image Translation with Deep Reinforcement Learning","date":"2023-09-24","arxiv_id":"2309.13672","repositories_listed":2,"syntology":null}],"record_sha256":"dbdea7daf2277b099fe1797382c4f15a08b5f0ce6522d8d9dfca5bdfd0d07c5c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}