{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/35","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":35,"pages_in_order":132,"rows_per_page":100,"rows":[3401,3500],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/34","next":"/task/reinforcement-learning/papers/36","papers":[{"url":"/paper/free-lunch-saliency-via-attention-in-atari","slug":"free-lunch-saliency-via-attention-in-atari","title":"Free-Lunch Saliency via Attention in Atari Agents","date":"2019-08-07","arxiv_id":"1908.02511","repositories_listed":1,"syntology":null},{"url":"/paper/doorgym-a-scalable-door-opening-environment","slug":"doorgym-a-scalable-door-opening-environment","title":"DoorGym: A Scalable Door Opening Environment And Baseline Agent","date":"2019-08-05","arxiv_id":"1908.01887","repositories_listed":1,"syntology":null},{"url":"/paper/dueling-posterior-sampling-for-preference","slug":"dueling-posterior-sampling-for-preference","title":"Dueling Posterior Sampling for Preference-Based Reinforcement Learning","date":"2019-08-04","arxiv_id":"1908.01289","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dueling-posterior-sampling-for-preference#ran","syntology_url":"https://syntology.ai/paper/1908.01289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.01289"}},"official":{"repos":["ernovoseller/DuelingPosteriorSampling"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/combining-learned-skills-and-reinforcement","slug":"combining-learned-skills-and-reinforcement","title":"Learning to combine primitive skills: A step towards versatile robotic manipulation","date":"2019-08-02","arxiv_id":"1908.00722","repositories_listed":1,"syntology":null},{"url":"/paper/health-informed-policy-gradients-for-multi","slug":"health-informed-policy-gradients-for-multi","title":"Health-Informed Policy Gradients for Multi-Agent Reinforcement Learning","date":"2019-08-02","arxiv_id":"1908.01022","repositories_listed":1,"syntology":null},{"url":"/paper/control-of-nonlinear-complex-and-black-boxed","slug":"control-of-nonlinear-complex-and-black-boxed","title":"Control of nonlinear, complex and black-boxed greenhouse system with reinforcement learning","date":"2019-07-30","arxiv_id":"1907.12690","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-adversarial-inverse-reinforcement","slug":"multi-agent-adversarial-inverse-reinforcement","title":"Multi-Agent Adversarial Inverse Reinforcement Learning","date":"2019-07-30","arxiv_id":"1907.13220","repositories_listed":1,"syntology":null},{"url":"/paper/reward-learning-for-efficient-reinforcement","slug":"reward-learning-for-efficient-reinforcement","title":"Reward Learning for Efficient Reinforcement Learning in Extractive Document Summarisation","date":"2019-07-30","arxiv_id":"1907.12894","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reward-learning-for-efficient-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1907.12894","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.12894"}},"official":{"repos":["UKPLab/ijcai2019-relis"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/action-grammars-a-cognitive-model-for","slug":"action-grammars-a-cognitive-model-for","title":"Semantic RL with Action Grammars: Data-Efficient Learning of Hierarchical Task Abstractions","date":"2019-07-29","arxiv_id":"1907.12477","repositories_listed":1,"syntology":null},{"url":"/paper/hindsight-trust-region-policy-optimization","slug":"hindsight-trust-region-policy-optimization","title":"Hindsight Trust Region Policy Optimization","date":"2019-07-29","arxiv_id":"1907.12439","repositories_listed":1,"syntology":null},{"url":"/paper/minerl-a-large-scale-dataset-of-minecraft","slug":"minerl-a-large-scale-dataset-of-minecraft","title":"MineRL: A Large-Scale Dataset of Minecraft Demonstrations","date":"2019-07-29","arxiv_id":"1907.13440","repositories_listed":1,"syntology":null},{"url":"/paper/reinforced-dynamic-reasoning-for-1","slug":"reinforced-dynamic-reasoning-for-1","title":"Reinforced Dynamic Reasoning for Conversational Question Generation","date":"2019-07-29","arxiv_id":"1907.12667","repositories_listed":1,"syntology":null},{"url":"/paper/making-sense-of-vision-and-touch-learning","slug":"making-sense-of-vision-and-touch-learning","title":"Making Sense of Vision and Touch: Learning Multimodal Representations for Contact-Rich Tasks","date":"2019-07-28","arxiv_id":"1907.13098","repositories_listed":1,"syntology":null},{"url":"/paper/self-imitation-learning-of-locomotion","slug":"self-imitation-learning-of-locomotion","title":"Self-Imitation Learning of Locomotion Movements through Termination Curriculum","date":"2019-07-27","arxiv_id":"1907.11842","repositories_listed":1,"syntology":null},{"url":"/paper/towards-model-based-reinforcement-learning","slug":"towards-model-based-reinforcement-learning","title":"Towards Model-based Reinforcement Learning for Industry-near Environments","date":"2019-07-27","arxiv_id":"1907.11971","repositories_listed":1,"syntology":null},{"url":"/paper/action-semantics-network-considering-the","slug":"action-semantics-network-considering-the","title":"Action Semantics Network: Considering the Effects of Actions in Multiagent Systems","date":"2019-07-26","arxiv_id":"1907.11461","repositories_listed":1,"syntology":null},{"url":"/paper/environment-probing-interaction-policies-1","slug":"environment-probing-interaction-policies-1","title":"Environment Probing Interaction Policies","date":"2019-07-26","arxiv_id":"1907.11740","repositories_listed":1,"syntology":null},{"url":"/paper/google-research-football-a-novel","slug":"google-research-football-a-novel","title":"Google Research Football: A Novel Reinforcement Learning Environment","date":"2019-07-25","arxiv_id":"1907.11180","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/google-research-football-a-novel#ran","syntology_url":"https://syntology.ai/paper/1907.11180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.11180"}},"official":{"repos":["google-research/football"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/discourse-marker-augmented-network-with-1","slug":"discourse-marker-augmented-network-with-1","title":"Discourse Marker Augmented Network with Reinforcement Learning for Natural Language Inference","date":"2019-07-23","arxiv_id":"1907.09692","repositories_listed":1,"syntology":null},{"url":"/paper/metalearned-neural-memory","slug":"metalearned-neural-memory","title":"Metalearned Neural Memory","date":"2019-07-23","arxiv_id":"1907.09720","repositories_listed":1,"syntology":null},{"url":"/paper/modeling-question-asking-using-neural-program","slug":"modeling-question-asking-using-neural-program","title":"Modeling question asking using neural program generation","date":"2019-07-23","arxiv_id":"1907.09899","repositories_listed":1,"syntology":null},{"url":"/paper/muscle-actuated-human-simulation-and-control","slug":"muscle-actuated-human-simulation-and-control","title":"Muscle-actuated Human Simulation and Control","date":"2019-07-23","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/structured-fusion-networks-for-dialog","slug":"structured-fusion-networks-for-dialog","title":"Structured Fusion Networks for Dialog","date":"2019-07-23","arxiv_id":"1907.10016","repositories_listed":1,"syntology":null},{"url":"/paper/characterizing-attacks-on-deep-reinforcement","slug":"characterizing-attacks-on-deep-reinforcement","title":"Characterizing Attacks on Deep Reinforcement Learning","date":"2019-07-21","arxiv_id":"1907.09470","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/characterizing-attacks-on-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1907.09470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.09470"}},"official":null}},{"url":"/paper/convolutional-reservoir-computing-for-world","slug":"convolutional-reservoir-computing-for-world","title":"Convolutional Reservoir Computing for World Models","date":"2019-07-18","arxiv_id":"1907.08040","repositories_listed":1,"syntology":null},{"url":"/paper/credit-assignment-as-a-proxy-for-transfer-in","slug":"credit-assignment-as-a-proxy-for-transfer-in","title":"Self-Attentional Credit Assignment for Transfer in Reinforcement Learning","date":"2019-07-18","arxiv_id":"1907.08027","repositories_listed":1,"syntology":null},{"url":"/paper/ppo-dash-improving-generalization-in-deep","slug":"ppo-dash-improving-generalization-in-deep","title":"PPO Dash: Improving Generalization in Deep Reinforcement Learning","date":"2019-07-15","arxiv_id":"1907.06704","repositories_listed":1,"syntology":null},{"url":"/paper/proximal-policy-optimization-with-mixed","slug":"proximal-policy-optimization-with-mixed","title":"Proximal Policy Optimization with Mixed Distributed Training","date":"2019-07-15","arxiv_id":"1907.06479","repositories_listed":1,"syntology":null},{"url":"/paper/finite-time-performance-bounds-and-adaptive","slug":"finite-time-performance-bounds-and-adaptive","title":"Finite-Time Performance Bounds and Adaptive Learning Rate Selection for Two Time-Scale Reinforcement Learning","date":"2019-07-14","arxiv_id":"1907.06290","repositories_listed":1,"syntology":null},{"url":"/paper/learning-self-correctable-policies-and-value","slug":"learning-self-correctable-policies-and-value","title":"Learning Self-Correctable Policies and Value Functions from Demonstrations with Negative Sampling","date":"2019-07-12","arxiv_id":"1907.05634","repositories_listed":1,"syntology":null},{"url":"/paper/effective-and-general-evaluation-for","slug":"effective-and-general-evaluation-for","title":"General Evaluation for Instruction Conditioned Navigation using Dynamic Time Warping","date":"2019-07-11","arxiv_id":"1907.05446","repositories_listed":1,"syntology":null},{"url":"/paper/deep-lagrangian-networks-for-end-to-end","slug":"deep-lagrangian-networks-for-end-to-end","title":"Deep Lagrangian Networks for end-to-end learning of energy-based control for under-actuated systems","date":"2019-07-10","arxiv_id":"1907.04489","repositories_listed":1,"syntology":null},{"url":"/paper/regularizing-neural-networks-for-future","slug":"regularizing-neural-networks-for-future","title":"Regularizing Neural Networks for Future Trajectory Prediction via Inverse Reinforcement Learning Framework","date":"2019-07-10","arxiv_id":"1907.04525","repositories_listed":1,"syntology":null},{"url":"/paper/striving-for-simplicity-in-off-policy-deep","slug":"striving-for-simplicity-in-off-policy-deep","title":"An Optimistic Perspective on Offline Reinforcement Learning","date":"2019-07-10","arxiv_id":"1907.04543","repositories_listed":1,"syntology":null},{"url":"/paper/learning-policies-for-social-network","slug":"learning-policies-for-social-network","title":"Influence maximization in unknown social networks: Learning Policies for Effective Graph Sampling","date":"2019-07-08","arxiv_id":"1907.11625","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":3,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/learning-policies-for-social-network#ran","syntology_url":"https://syntology.ai/paper/1907.11625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.11625"}},"official":{"repos":["kage08/graph_sample_rl"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/hardwaresoftware-co-exploration-of-neural","slug":"hardwaresoftware-co-exploration-of-neural","title":"Hardware/Software Co-Exploration of Neural Architectures","date":"2019-07-06","arxiv_id":"1907.04650","repositories_listed":1,"syntology":null},{"url":"/paper/attentive-multi-task-deep-reinforcement","slug":"attentive-multi-task-deep-reinforcement","title":"Attentive Multi-Task Deep Reinforcement Learning","date":"2019-07-05","arxiv_id":"1907.02874","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/attentive-multi-task-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1907.02874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.02874"}},"official":{"repos":["braemt/attentive-multi-task-deep-reinforcement-learning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/co-training-for-policy-learning","slug":"co-training-for-policy-learning","title":"Co-training for Policy Learning","date":"2019-07-03","arxiv_id":"1907.04484","repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-and-generalization-in-rl-a-tool-use","slug":"reasoning-and-generalization-in-rl-a-tool-use","title":"Reasoning and Generalization in RL: A Tool Use Perspective","date":"2019-07-03","arxiv_id":"1907.02050","repositories_listed":1,"syntology":null},{"url":"/paper/conservative-q-improvement-reinforcement","slug":"conservative-q-improvement-reinforcement","title":"Conservative Q-Improvement: Reinforcement Learning for an Interpretable Decision-Tree Policy","date":"2019-07-02","arxiv_id":"1907.01180","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conservative-q-improvement-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1907.01180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.01180"}},"official":null}},{"url":"/paper/a-deep-reinforced-sequence-to-set-model-for-1","slug":"a-deep-reinforced-sequence-to-set-model-for-1","title":"A Deep Reinforced Sequence-to-Set Model for Multi-Label Classification","date":"2019-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-how-to-active-learn-by-dreaming","slug":"learning-how-to-active-learn-by-dreaming","title":"Learning How to Active Learn by Dreaming","date":"2019-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reinforced-training-data-selection-for-domain","slug":"reinforced-training-data-selection-for-domain","title":"Reinforced Training Data Selection for Domain Adaptation","date":"2019-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/towards-fine-grained-text-sentiment-transfer","slug":"towards-fine-grained-text-sentiment-transfer","title":"Towards Fine-grained Text Sentiment Transfer","date":"2019-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/detecting-spiky-corruption-in-markov-decision","slug":"detecting-spiky-corruption-in-markov-decision","title":"Detecting Spiky Corruption in Markov Decision Processes","date":"2019-06-30","arxiv_id":"1907.00452","repositories_listed":1,"syntology":null},{"url":"/paper/multiple-landmark-detection-using-multi-agent","slug":"multiple-landmark-detection-using-multi-agent","title":"Multiple Landmark Detection using Multi-Agent Reinforcement Learning","date":"2019-06-30","arxiv_id":"1907.00318","repositories_listed":1,"syntology":null},{"url":"/paper/variational-quantum-circuits-and-deep","slug":"variational-quantum-circuits-and-deep","title":"Variational Quantum Circuits for Deep Reinforcement Learning","date":"2019-06-30","arxiv_id":"1907.00397","repositories_listed":1,"syntology":null},{"url":"/paper/way-off-policy-batch-deep-reinforcement","slug":"way-off-policy-batch-deep-reinforcement","title":"Way Off-Policy Batch Deep Reinforcement Learning of Implicit Human Preferences in Dialog","date":"2019-06-30","arxiv_id":"1907.00456","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/way-off-policy-batch-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1907.00456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.00456"}},"official":{"repos":["natashamjaques/neural_chat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/growing-action-spaces","slug":"growing-action-spaces","title":"Growing Action Spaces","date":"2019-06-28","arxiv_id":"1906.12266","repositories_listed":1,"syntology":null},{"url":"/paper/emergence-of-exploratory-look-around-1","slug":"emergence-of-exploratory-look-around-1","title":"Emergence of Exploratory Look-Around Behaviors through Active Observation Completion","date":"2019-06-27","arxiv_id":"1906.11407","repositories_listed":1,"syntology":null},{"url":"/paper/hyp-rl-hyperparameter-optimization-by","slug":"hyp-rl-hyperparameter-optimization-by","title":"Hyp-RL : Hyperparameter Optimization by Reinforcement Learning","date":"2019-06-27","arxiv_id":"1906.11527","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hyp-rl-hyperparameter-optimization-by#ran","syntology_url":"https://syntology.ai/paper/1906.11527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.11527"}},"official":{"repos":["hadijomaa/HypRL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-based-model-predictive-control-for-1","slug":"learning-based-model-predictive-control-for-1","title":"Learning-based Model Predictive Control for Safe Exploration and Reinforcement Learning","date":"2019-06-27","arxiv_id":"1906.12189","repositories_listed":1,"syntology":null},{"url":"/paper/cooperation-aware-reinforcement-learning-for","slug":"cooperation-aware-reinforcement-learning-for","title":"Cooperation-Aware Reinforcement Learning for Merging in Dense Traffic","date":"2019-06-26","arxiv_id":"1906.11021","repositories_listed":1,"syntology":null},{"url":"/paper/pyrep-bringing-v-rep-to-deep-robot-learning","slug":"pyrep-bringing-v-rep-to-deep-robot-learning","title":"PyRep: Bringing V-REP to Deep Robot Learning","date":"2019-06-26","arxiv_id":"1906.11176","repositories_listed":1,"syntology":null},{"url":"/paper/towards-empathic-deep-q-learning","slug":"towards-empathic-deep-q-learning","title":"Towards Empathic Deep Q-Learning","date":"2019-06-26","arxiv_id":"1906.10918","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-empathic-deep-q-learning#ran","syntology_url":"https://syntology.ai/paper/1906.10918","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.10918"}},"official":{"repos":["bartbussmann/EmpathicDQN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/proximal-distilled-evolutionary-reinforcement","slug":"proximal-distilled-evolutionary-reinforcement","title":"Proximal Distilled Evolutionary Reinforcement Learning","date":"2019-06-24","arxiv_id":"1906.09807","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/proximal-distilled-evolutionary-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1906.09807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.09807"}},"official":{"repos":["crisbodnar/pderl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ranking-policy-gradient","slug":"ranking-policy-gradient","title":"Ranking Policy Gradient","date":"2019-06-24","arxiv_id":"1906.09674","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ranking-policy-gradient#ran","syntology_url":"https://syntology.ai/paper/1906.09674","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.09674"}},"official":{"repos":["illidanlab/rpg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-neurally-plausible-model-learns-successor","slug":"a-neurally-plausible-model-learns-successor","title":"A neurally plausible model learns successor representations in partially observable environments","date":"2019-06-22","arxiv_id":"1906.09480","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/a-neurally-plausible-model-learns-successor#ran","syntology_url":"https://syntology.ai/paper/1906.09480","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.09480"}},"official":{"repos":["evertes/distributional_SF"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/learning-belief-representations-for-imitation","slug":"learning-belief-representations-for-imitation","title":"Learning Belief Representations for Imitation Learning in POMDPs","date":"2019-06-22","arxiv_id":"1906.09510","repositories_listed":1,"syntology":null},{"url":"/paper/learning-reward-functions-by-integrating","slug":"learning-reward-functions-by-integrating","title":"Learning Reward Functions by Integrating Human Demonstrations and Preferences","date":"2019-06-21","arxiv_id":"1906.08928","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-models-of-human","slug":"reinforcement-learning-models-of-human","title":"A Story of Two Streams: Reinforcement Learning Models from Human Behavior and Neuropsychiatry","date":"2019-06-21","arxiv_id":"1906.11286","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-convex","slug":"reinforcement-learning-with-convex","title":"Reinforcement Learning with Convex Constraints","date":"2019-06-21","arxiv_id":"1906.09323","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-convex#ran","syntology_url":"https://syntology.ai/paper/1906.09323","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.09323"}},"official":{"repos":["xkianteb/ApproPO"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/split-q-learning-reinforcement-learning-with","slug":"split-q-learning-reinforcement-learning-with","title":"Split Q Learning: Reinforcement Learning with Two-Stream Rewards","date":"2019-06-21","arxiv_id":"1906.12350","repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-reinforcement-learning-approach-for-1","slug":"a-deep-reinforcement-learning-approach-for-1","title":"A Deep Reinforcement Learning Approach for Global Routing","date":"2019-06-20","arxiv_id":"1906.08809","repositories_listed":1,"syntology":null},{"url":"/paper/a-hierarchical-architecture-for-sequential","slug":"a-hierarchical-architecture-for-sequential","title":"A Hierarchical Architecture for Sequential Decision-Making in Autonomous Driving using Deep Reinforcement Learning","date":"2019-06-20","arxiv_id":"1906.08464","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-model-based-planning-with-policy","slug":"exploring-model-based-planning-with-policy","title":"Exploring Model-based Planning with Policy Networks","date":"2019-06-20","arxiv_id":"1906.08649","repositories_listed":1,"syntology":null},{"url":"/paper/calibrated-model-based-deep-reinforcement","slug":"calibrated-model-based-deep-reinforcement","title":"Calibrated Model-Based Deep Reinforcement Learning","date":"2019-06-19","arxiv_id":"1906.08312","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-plan-hierarchically-from","slug":"learning-to-plan-hierarchically-from","title":"Learning to Plan Hierarchically from Curriculum","date":"2019-06-18","arxiv_id":"1906.07371","repositories_listed":1,"syntology":null},{"url":"/paper/learning-driven-exploration-for-reinforcement","slug":"learning-driven-exploration-for-reinforcement","title":"Learning-Driven Exploration for Reinforcement Learning","date":"2019-06-17","arxiv_id":"1906.06890","repositories_listed":1,"syntology":null},{"url":"/paper/curriculum-learning-for-cumulative-return","slug":"curriculum-learning-for-cumulative-return","title":"Curriculum Learning for Cumulative Return Maximization","date":"2019-06-13","arxiv_id":"1906.06178","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-industrial","slug":"deep-reinforcement-learning-for-industrial","title":"Deep Reinforcement Learning for Industrial Insertion Tasks with Visual Inputs and Natural Rewards","date":"2019-06-13","arxiv_id":"1906.05841","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/deep-reinforcement-learning-for-industrial#ran","syntology_url":"https://syntology.ai/paper/1906.05841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.05841"}},"official":null}},{"url":"/paper/goal-conditioned-imitation-learning","slug":"goal-conditioned-imitation-learning","title":"Goal-conditioned Imitation Learning","date":"2019-06-13","arxiv_id":"1906.05838","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/goal-conditioned-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/1906.05838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.05838"}},"official":{"repos":["dingyiming0427/goalgail"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-and-accurate-estimation-of","slug":"efficient-and-accurate-estimation-of","title":"Efficient and Accurate Estimation of Lipschitz Constants for Deep Neural Networks","date":"2019-06-12","arxiv_id":"1906.04893","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-exploration-via-state-marginal","slug":"efficient-exploration-via-state-marginal","title":"Efficient Exploration via State Marginal Matching","date":"2019-06-12","arxiv_id":"1906.05274","repositories_listed":1,"syntology":null},{"url":"/paper/meta-learning-via-learned-loss","slug":"meta-learning-via-learned-loss","title":"Meta-Learning via Learned Loss","date":"2019-06-12","arxiv_id":"1906.05374","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-knowledge-graph-reasoning-for","slug":"reinforcement-knowledge-graph-reasoning-for","title":"Reinforcement Knowledge Graph Reasoning for Explainable Recommendation","date":"2019-06-12","arxiv_id":"1906.05237","repositories_listed":1,"syntology":null},{"url":"/paper/search-on-the-replay-buffer-bridging-planning","slug":"search-on-the-replay-buffer-bridging-planning","title":"Search on the Replay Buffer: Bridging Planning and Reinforcement Learning","date":"2019-06-12","arxiv_id":"1906.05253","repositories_listed":1,"syntology":null},{"url":"/paper/causal-discovery-with-reinforcement-learning","slug":"causal-discovery-with-reinforcement-learning","title":"Causal Discovery with Reinforcement Learning","date":"2019-06-11","arxiv_id":"1906.04477","repositories_listed":1,"syntology":null},{"url":"/paper/learning-powerful-policies-by-using","slug":"learning-powerful-policies-by-using","title":"Learning Powerful Policies by Using Consistent Dynamics Model","date":"2019-06-11","arxiv_id":"1906.04355","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-channel-coding","slug":"reinforcement-learning-for-channel-coding","title":"Reinforcement Learning for Channel Coding: Learned Bit-Flipping Decoding","date":"2019-06-11","arxiv_id":"1906.04448","repositories_listed":1,"syntology":null},{"url":"/paper/wasserstein-reinforcement-learning","slug":"wasserstein-reinforcement-learning","title":"Learning to Score Behaviors for Guided Policy Optimization","date":"2019-06-11","arxiv_id":"1906.04349","repositories_listed":1,"syntology":null},{"url":"/paper/weight-agnostic-neural-networks","slug":"weight-agnostic-neural-networks","title":"Weight Agnostic Neural Networks","date":"2019-06-11","arxiv_id":"1906.04358","repositories_listed":1,"syntology":null},{"url":"/paper/exploration-via-hindsight-goal-generation","slug":"exploration-via-hindsight-goal-generation","title":"Exploration via Hindsight Goal Generation","date":"2019-06-10","arxiv_id":"1906.04279","repositories_listed":1,"syntology":null},{"url":"/paper/neural-keyphrase-generation-via-reinforcement","slug":"neural-keyphrase-generation-via-reinforcement","title":"Neural Keyphrase Generation via Reinforcement Learning with Adaptive Rewards","date":"2019-06-10","arxiv_id":"1906.04106","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/neural-keyphrase-generation-via-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1906.04106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.04106"}},"official":{"repos":["kenchan0226/keyphrase-generation-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-amortized-ranking-critical-training","slug":"towards-amortized-ranking-critical-training","title":"Towards Amortized Ranking-Critical Training for Collaborative Filtering","date":"2019-06-10","arxiv_id":"1906.04281","repositories_listed":1,"syntology":null},{"url":"/paper/curiosity-driven-multi-criteria-hindsight","slug":"curiosity-driven-multi-criteria-hindsight","title":"Curiosity-Driven Multi-Criteria Hindsight Experience Replay","date":"2019-06-09","arxiv_id":"1906.03710","repositories_listed":1,"syntology":null},{"url":"/paper/gossip-based-actor-learner-architectures-for","slug":"gossip-based-actor-learner-architectures-for","title":"Gossip-based Actor-Learner Architectures for Deep Reinforcement Learning","date":"2019-06-09","arxiv_id":"1906.04585","repositories_listed":1,"syntology":null},{"url":"/paper/svrg-for-policy-evaluation-with-fewer","slug":"svrg-for-policy-evaluation-with-fewer","title":"SVRG for Policy Evaluation with Fewer Gradient Evaluations","date":"2019-06-09","arxiv_id":"1906.03704","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/svrg-for-policy-evaluation-with-fewer#ran","syntology_url":"https://syntology.ai/paper/1906.03704","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.03704"}},"official":null}},{"url":"/paper/ego-pose-estimation-and-forecasting-as-real","slug":"ego-pose-estimation-and-forecasting-as-real","title":"Ego-Pose Estimation and Forecasting as Real-Time PD Control","date":"2019-06-07","arxiv_id":"1906.03173","repositories_listed":1,"syntology":null},{"url":"/paper/preference-based-interactive-multi-document","slug":"preference-based-interactive-multi-document","title":"Preference-based Interactive Multi-Document Summarisation","date":"2019-06-07","arxiv_id":"1906.02923","repositories_listed":1,"syntology":null},{"url":"/paper/improving-exploration-in-soft-actor-critic","slug":"improving-exploration-in-soft-actor-critic","title":"Improving Exploration in Soft-Actor-Critic with Normalizing Flows Policies","date":"2019-06-06","arxiv_id":"1906.02771","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-exploration-in-soft-actor-critic#ran","syntology_url":"https://syntology.ai/paper/1906.02771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.02771"}},"official":{"repos":["joeybose/FloRL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-interpretable-reinforcement-learning","slug":"towards-interpretable-reinforcement-learning","title":"Towards Interpretable Reinforcement Learning Using Attention Augmented Agents","date":"2019-06-06","arxiv_id":"1906.02500","repositories_listed":1,"syntology":null},{"url":"/paper/an-imitation-learning-approach-to","slug":"an-imitation-learning-approach-to","title":"An Imitation Learning Approach to Unsupervised Parsing","date":"2019-06-05","arxiv_id":"1906.02276","repositories_listed":1,"syntology":null},{"url":"/paper/finding-friend-and-foe-in-multi-agent-games","slug":"finding-friend-and-foe-in-multi-agent-games","title":"Finding Friend and Foe in Multi-Agent Games","date":"2019-06-05","arxiv_id":"1906.02330","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-when-all-actions-are","slug":"reinforcement-learning-when-all-actions-are","title":"Reinforcement Learning When All Actions are Not Always Available","date":"2019-06-05","arxiv_id":"1906.01772","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-low-complexity","slug":"reinforcement-learning-with-low-complexity","title":"Reinforcement Learning with Low-Complexity Liquid State Machines","date":"2019-06-04","arxiv_id":"1906.01695","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reinforcement-learning-with-low-complexity#ran","syntology_url":"https://syntology.ai/paper/1906.01695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.01695"}},"official":{"repos":["wponghiran/lsm-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-exploration-in-linear-quadratic","slug":"robust-exploration-in-linear-quadratic","title":"Robust exploration in linear quadratic reinforcement learning","date":"2019-06-04","arxiv_id":"1906.01584","repositories_listed":1,"syntology":null},{"url":"/paper/190600549","slug":"190600549","title":"Know More about Each Other: Evolving Dialogue Strategy via Compound Assessment","date":"2019-06-03","arxiv_id":"1906.00549","repositories_listed":1,"syntology":null},{"url":"/paper/190600584","slug":"190600584","title":"A Semi-Supervised Approach for Low-Resourced Text Generation","date":"2019-06-03","arxiv_id":"1906.00584","repositories_listed":1,"syntology":null},{"url":"/paper/190600889","slug":"190600889","title":"Learning to solve the credit assignment problem","date":"2019-06-03","arxiv_id":"1906.00889","repositories_listed":1,"syntology":null}],"record_sha256":"1f7467d3ba32192e4b56f2863a5ce2e71ea195a23ed121cf70c068c03798f2b5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}