{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/34","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":34,"pages_in_order":132,"rows_per_page":100,"rows":[3301,3400],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/33","next":"/task/reinforcement-learning/papers/35","papers":[{"url":"/paper/single-episode-policy-transfer-in","slug":"single-episode-policy-transfer-in","title":"Single Episode Policy Transfer in Reinforcement Learning","date":"2019-10-17","arxiv_id":"1910.07719","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/single-episode-policy-transfer-in#ran","syntology_url":"https://syntology.ai/paper/1910.07719","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.07719"}},"official":{"repos":["011235813/SEPT"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/deep-reinforcement-learning-meets-graph","slug":"deep-reinforcement-learning-meets-graph","title":"Deep Reinforcement Learning meets Graph Neural Networks: exploring a routing optimization use case","date":"2019-10-16","arxiv_id":"1910.07421","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-meets-graph#ran","syntology_url":"https://syntology.ai/paper/1910.07421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.07421"}},"official":{"repos":["knowledgedefinednetworking/DRL-GNN"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/approximate-inference-in-discrete","slug":"approximate-inference-in-discrete","title":"Approximate Inference in Discrete Distributions with Monte Carlo Tree Search and Value Functions","date":"2019-10-15","arxiv_id":"1910.06862","repositories_listed":1,"syntology":null},{"url":"/paper/derivative-free-optimization-of-neural","slug":"derivative-free-optimization-of-neural","title":"Derivative-Free Optimization of Neural Networks using Local Search","date":"2019-10-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-classifiers-on-positive-and","slug":"learning-classifiers-on-positive-and","title":"Learning Classifiers on Positive and Unlabeled Data with Policy Gradient","date":"2019-10-15","arxiv_id":"1910.06535","repositories_listed":1,"syntology":null},{"url":"/paper/model-free-reinforcement-learning-in-infinite","slug":"model-free-reinforcement-learning-in-infinite","title":"Model-free Reinforcement Learning in Infinite-horizon Average-reward Markov Decision Processes","date":"2019-10-15","arxiv_id":"1910.07072","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-spiking-coagents","slug":"reinforcement-learning-with-spiking-coagents","title":"Reinforcement learning with a network of spiking agents","date":"2019-10-15","arxiv_id":"1910.06489","repositories_listed":1,"syntology":null},{"url":"/paper/bootstrapping-the-expressivity-with-model","slug":"bootstrapping-the-expressivity-with-model","title":"On the Expressivity of Neural Networks for Deep Reinforcement Learning","date":"2019-10-14","arxiv_id":"1910.05927","repositories_listed":1,"syntology":null},{"url":"/paper/policy-poisoning-in-batch-reinforcement","slug":"policy-poisoning-in-batch-reinforcement","title":"Policy Poisoning in Batch Reinforcement Learning and Control","date":"2019-10-13","arxiv_id":"1910.05821","repositories_listed":1,"syntology":null},{"url":"/paper/autonomous-navigation-via-deep-reinforcement","slug":"autonomous-navigation-via-deep-reinforcement","title":"Autonomous Navigation via Deep Reinforcement Learning for Resource Constraint Edge Nodes using Transfer Learning","date":"2019-10-12","arxiv_id":"1910.05547","repositories_listed":1,"syntology":null},{"url":"/paper/influence-based-multi-agent-exploration","slug":"influence-based-multi-agent-exploration","title":"Influence-Based Multi-Agent Exploration","date":"2019-10-12","arxiv_id":"1910.05512","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/influence-based-multi-agent-exploration#ran","syntology_url":"https://syntology.ai/paper/1910.05512","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.05512"}},"official":{"repos":["TonghanWang/EITI-EDTI"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-nearly-decomposable-value-functions-1","slug":"learning-nearly-decomposable-value-functions-1","title":"Learning Nearly Decomposable Value Functions Via Communication Minimization","date":"2019-10-11","arxiv_id":"1910.05366","repositories_listed":1,"syntology":null},{"url":"/paper/agent-with-warm-start-and-active-termination","slug":"agent-with-warm-start-and-active-termination","title":"Agent with Warm Start and Active Termination for Plane Localization in 3D Ultrasound","date":"2019-10-10","arxiv_id":"1910.04331","repositories_listed":1,"syntology":null},{"url":"/paper/from-visual-place-recognition-to-navigation","slug":"from-visual-place-recognition-to-navigation","title":"CityLearn: Diverse Real-World Environments for Sample-Efficient Navigation Policy Learning","date":"2019-10-10","arxiv_id":"1910.04335","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-reinforcement-learning-with-3","slug":"hierarchical-reinforcement-learning-with-3","title":"Hierarchical Reinforcement Learning with Advantage-Based Auxiliary Rewards","date":"2019-10-10","arxiv_id":"1910.04450","repositories_listed":1,"syntology":null},{"url":"/paper/neuro-dram-a-3d-recurrent-visual-attention","slug":"neuro-dram-a-3d-recurrent-visual-attention","title":"NEURO-DRAM: a 3D recurrent visual attention model for interpretable neuroimaging classification","date":"2019-10-10","arxiv_id":"1910.04721","repositories_listed":1,"syntology":null},{"url":"/paper/read-highlight-and-summarize-a-hierarchical","slug":"read-highlight-and-summarize-a-hierarchical","title":"Read, Highlight and Summarize: A Hierarchical Neural Semantic Encoder-based Approach","date":"2019-10-08","arxiv_id":"1910.03177","repositories_listed":1,"syntology":null},{"url":"/paper/self-paced-contextual-reinforcement-learning","slug":"self-paced-contextual-reinforcement-learning","title":"Self-Paced Contextual Reinforcement Learning","date":"2019-10-07","arxiv_id":"1910.02826","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-paced-contextual-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1910.02826","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.02826"}},"official":{"repos":["psclklnk/self-paced-rl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-q-network-for-angry-birds","slug":"deep-q-network-for-angry-birds","title":"Deep Q-Network for Angry Birds","date":"2019-10-04","arxiv_id":"1910.01806","repositories_listed":1,"syntology":null},{"url":"/paper/quantized-reinforcement-learning-quarl","slug":"quantized-reinforcement-learning-quarl","title":"QuaRL: Quantization for Fast and Environmentally Sustainable Reinforcement Learning","date":"2019-10-02","arxiv_id":"1910.01055","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quantized-reinforcement-learning-quarl#ran","syntology_url":"https://syntology.ai/paper/1910.01055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.01055"}},"official":{"repos":["harvard-edge/quarl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-doodling-and-painting-with","slug":"unsupervised-doodling-and-painting-with","title":"Unsupervised Doodling and Painting with Improved SPIRAL","date":"2019-10-02","arxiv_id":"1910.01007","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-bimanual-manipulation-using-learned","slug":"efficient-bimanual-manipulation-using-learned","title":"Efficient Bimanual Manipulation Using Learned Task Schemas","date":"2019-09-30","arxiv_id":"1909.13874","repositories_listed":1,"syntology":null},{"url":"/paper/hamiltonian-generative-networks","slug":"hamiltonian-generative-networks","title":"Hamiltonian Generative Networks","date":"2019-09-30","arxiv_id":"1909.13789","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hamiltonian-generative-networks#ran","syntology_url":"https://syntology.ai/paper/1909.13789","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.13789"}},"official":null}},{"url":"/paper/multiagent-rollout-algorithms-and","slug":"multiagent-rollout-algorithms-and","title":"Multiagent Rollout Algorithms and Reinforcement Learning","date":"2019-09-30","arxiv_id":"1910.00120","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-evaluate-machine-learning-approaches","slug":"how-to-evaluate-machine-learning-approaches","title":"How to Evaluate Machine Learning Approaches for Combinatorial Optimization: Application to the Travelling Salesman Problem","date":"2019-09-28","arxiv_id":"1909.13121","repositories_listed":1,"syntology":null},{"url":"/paper/relational-graph-learning-for-crowd","slug":"relational-graph-learning-for-crowd","title":"Relational Graph Learning for Crowd Navigation","date":"2019-09-28","arxiv_id":"1909.13165","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-roi-generation-for-video-object","slug":"adaptive-roi-generation-for-video-object","title":"Adaptive ROI Generation for Video Object Segmentation Using Reinforcement Learning","date":"2019-09-27","arxiv_id":"1909.12482","repositories_listed":1,"syntology":null},{"url":"/paper/mdp-based-shallow-parsing-in-distantly","slug":"mdp-based-shallow-parsing-in-distantly","title":"Distantly Supervised Question Parsing","date":"2019-09-27","arxiv_id":"1909.12566","repositories_listed":1,"syntology":null},{"url":"/paper/a-framework-for-data-driven-robotics","slug":"a-framework-for-data-driven-robotics","title":"Scaling data-driven robotics with reward sketching and batch reinforcement learning","date":"2019-09-26","arxiv_id":"1909.12200","repositories_listed":1,"syntology":null},{"url":"/paper/a-simulation-of-uav-power-optimization-via","slug":"a-simulation-of-uav-power-optimization-via","title":"Visual Exploration and Energy-aware Path Planning via Reinforcement Learning","date":"2019-09-26","arxiv_id":"1909.12217","repositories_listed":1,"syntology":null},{"url":"/paper/harnessing-structures-for-value-based","slug":"harnessing-structures-for-value-based","title":"Harnessing Structures for Value-Based Planning and Reinforcement Learning","date":"2019-09-26","arxiv_id":"1909.12255","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/harnessing-structures-for-value-based#ran","syntology_url":"https://syntology.ai/paper/1909.12255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.12255"}},"official":{"repos":["YyzHarry/SV-RL"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rl-lim-reinforcement-learning-based-locally","slug":"rl-lim-reinforcement-learning-based-locally","title":"LIMIS: Locally Interpretable Modeling using Instance-wise Subsampling","date":"2019-09-26","arxiv_id":"1909.12367","repositories_listed":1,"syntology":null},{"url":"/paper/v-mpo-on-policy-maximum-a-posteriori-policy","slug":"v-mpo-on-policy-maximum-a-posteriori-policy","title":"V-MPO: On-Policy Maximum a Posteriori Policy Optimization for Discrete and Continuous Control","date":"2019-09-26","arxiv_id":"1909.12238","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/v-mpo-on-policy-maximum-a-posteriori-policy#ran","syntology_url":"https://syntology.ai/paper/1909.12238","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.12238"}},"official":null}},{"url":"/paper/good-robot-efficient-reinforcement-learning","slug":"good-robot-efficient-reinforcement-learning","title":"\"Good Robot!\": Efficient Reinforcement Learning for Multi-Step Visual Tasks with Sim to Real Transfer","date":"2019-09-25","arxiv_id":"1909.11730","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-seek-autonomous-source-seeking","slug":"learning-to-seek-autonomous-source-seeking","title":"Learning to Seek: Autonomous Source Seeking with Deep Reinforcement Learning Onboard a Nano Drone Microcontroller","date":"2019-09-25","arxiv_id":"1909.11236","repositories_listed":1,"syntology":null},{"url":"/paper/robel-robotics-benchmarks-for-learning-with","slug":"robel-robotics-benchmarks-for-learning-with","title":"ROBEL: Robotics Benchmarks for Learning with Low-Cost Robots","date":"2019-09-25","arxiv_id":"1909.11639","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-state-control-through","slug":"self-supervised-state-control-through","title":"Self-Supervised State-Control through Intrinsic Mutual Information Rewards","date":"2019-09-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/accept-synthetic-objects-as-real-end-to-end","slug":"accept-synthetic-objects-as-real-end-to-end","title":"Accept Synthetic Objects as Real: End-to-End Training of Attentive Deep Visuomotor Policies for Manipulation in Clutter","date":"2019-09-24","arxiv_id":"1909.11128","repositories_listed":1,"syntology":null},{"url":"/paper/avoidance-learning-using-observational","slug":"avoidance-learning-using-observational","title":"Avoidance Learning Using Observational Reinforcement Learning","date":"2019-09-24","arxiv_id":"1909.11228","repositories_listed":1,"syntology":null},{"url":"/paper/demystifying-active-inference","slug":"demystifying-active-inference","title":"Active inference: demystified and compared","date":"2019-09-24","arxiv_id":"1909.10863","repositories_listed":1,"syntology":null},{"url":"/paper/invariant-transform-experience-replay","slug":"invariant-transform-experience-replay","title":"Invariant Transform Experience Replay: Data Augmentation for Deep Reinforcement Learning","date":"2019-09-24","arxiv_id":"1909.10707","repositories_listed":1,"syntology":null},{"url":"/paper/paying-attention-to-function-words","slug":"paying-attention-to-function-words","title":"Paying Attention to Function Words","date":"2019-09-24","arxiv_id":"1909.11060","repositories_listed":1,"syntology":null},{"url":"/paper/190910470","slug":"190910470","title":"Improving Generative Visual Dialog by Answering Diverse Questions","date":"2019-09-23","arxiv_id":"1909.10470","repositories_listed":1,"syntology":null},{"url":"/paper/loaded-dice-trading-off-bias-and-variance-in","slug":"loaded-dice-trading-off-bias-and-variance-in","title":"Loaded DiCE: Trading off Bias and Variance in Any-Order Score Function Estimators for Reinforcement Learning","date":"2019-09-23","arxiv_id":"1909.10549","repositories_listed":1,"syntology":null},{"url":"/paper/190910008","slug":"190910008","title":"Multi-task Learning and Catastrophic Forgetting in Continual Reinforcement Learning","date":"2019-09-22","arxiv_id":"1909.10008","repositories_listed":1,"syntology":null},{"url":"/paper/190909902","slug":"190909902","title":"Deep Reinforcement Learning with Modulated Hebbian plus Q Network Architecture","date":"2019-09-21","arxiv_id":"1909.09902","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-optimization-for-iterative-learning","slug":"bayesian-optimization-for-iterative-learning","title":"Bayesian Optimization for Iterative Learning","date":"2019-09-20","arxiv_id":"1909.09593","repositories_listed":1,"syntology":null},{"url":"/paper/meta-inverse-reinforcement-learning-with","slug":"meta-inverse-reinforcement-learning-with","title":"Meta-Inverse Reinforcement Learning with Probabilistic Context Variables","date":"2019-09-20","arxiv_id":"1909.09314","repositories_listed":1,"syntology":null},{"url":"/paper/modelicagym-applying-reinforcement-learning","slug":"modelicagym-applying-reinforcement-learning","title":"ModelicaGym: Applying Reinforcement Learning to Modelica Models","date":"2019-09-18","arxiv_id":"1909.08604","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-policy-gradient-methods-with","slug":"sample-efficient-policy-gradient-methods-with","title":"Sample Efficient Policy Gradient Methods with Recursive Variance Reduction","date":"2019-09-18","arxiv_id":"1909.08610","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-reinforcement-learning-for-open","slug":"hierarchical-reinforcement-learning-for-open","title":"Hierarchical Reinforcement Learning for Open-Domain Dialog","date":"2019-09-17","arxiv_id":"1909.07547","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-manipulate-object-collections","slug":"learning-to-manipulate-object-collections","title":"Learning to Manipulate Object Collections Using Grounded State Representations","date":"2019-09-17","arxiv_id":"1909.07876","repositories_listed":1,"syntology":null},{"url":"/paper/mdp-playground-meta-features-in-reinforcement","slug":"mdp-playground-meta-features-in-reinforcement","title":"MDP Playground: An Analysis and Debug Testbed for Reinforcement Learning","date":"2019-09-17","arxiv_id":"1909.07750","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mdp-playground-meta-features-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1909.07750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.07750"}},"official":{"repos":["automl/mdp-playground"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/control-synthesis-from-linear-temporal-logic","slug":"control-synthesis-from-linear-temporal-logic","title":"Control Synthesis from Linear Temporal Logic Specifications using Model-Free Reinforcement Learning","date":"2019-09-16","arxiv_id":"1909.07299","repositories_listed":1,"syntology":null},{"url":"/paper/domain-transfer-in-dialogue-systems-without","slug":"domain-transfer-in-dialogue-systems-without","title":"Domain Transfer in Dialogue Systems without Turn-Level Supervision","date":"2019-09-16","arxiv_id":"1909.07101","repositories_listed":1,"syntology":null},{"url":"/paper/a-neural-approach-to-irony-generation","slug":"a-neural-approach-to-irony-generation","title":"A Neural Approach to Irony Generation","date":"2019-09-13","arxiv_id":"1909.06200","repositories_listed":1,"syntology":null},{"url":"/paper/dl2-a-deep-learning-driven-scheduler-for-deep","slug":"dl2-a-deep-learning-driven-scheduler-for-deep","title":"DL2: A Deep Learning-driven Scheduler for Deep Learning Clusters","date":"2019-09-13","arxiv_id":"1909.06040","repositories_listed":1,"syntology":null},{"url":"/paper/isl-optimal-policy-learning-with-optimal","slug":"isl-optimal-policy-learning-with-optimal","title":"ISL: A novel approach for deep exploration","date":"2019-09-13","arxiv_id":"1909.06293","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-portfolio","slug":"reinforcement-learning-for-portfolio","title":"Reinforcement Learning for Portfolio Management","date":"2019-09-12","arxiv_id":"1909.09571","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-learning-and-exploration-of","slug":"unsupervised-learning-and-exploration-of","title":"Unsupervised Learning and Exploration of Reachable Outcome Space","date":"2019-09-12","arxiv_id":"1909.05508","repositories_listed":1,"syntology":null},{"url":"/paper/predicting-optimal-value-functions-by","slug":"predicting-optimal-value-functions-by","title":"Predicting optimal value functions by interpolating reward functions in scalarized multi-objective reinforcement learning","date":"2019-09-11","arxiv_id":"1909.05004","repositories_listed":1,"syntology":null},{"url":"/paper/recsim-a-configurable-simulation-platform-for","slug":"recsim-a-configurable-simulation-platform-for","title":"RecSim: A Configurable Simulation Platform for Recommender Systems","date":"2019-09-11","arxiv_id":"1909.04847","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/recsim-a-configurable-simulation-platform-for#ran","syntology_url":"https://syntology.ai/paper/1909.04847","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.04847"}},"official":{"repos":["google-research/recsim"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-temporal-logic","slug":"reinforcement-learning-for-temporal-logic","title":"Reinforcement Learning for Temporal Logic Control Synthesis with Probabilistic Satisfaction Guarantees","date":"2019-09-11","arxiv_id":"1909.05304","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-for-temporal-logic#ran","syntology_url":"https://syntology.ai/paper/1909.05304","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.05304"}},"official":{"repos":["grockious/lcrl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/what-makes-a-good-story-designing-composite","slug":"what-makes-a-good-story-designing-composite","title":"What Makes A Good Story? Designing Composite Rewards for Visual Storytelling","date":"2019-09-11","arxiv_id":"1909.05316","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/what-makes-a-good-story-designing-composite#ran","syntology_url":"https://syntology.ai/paper/1909.05316","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.05316"}},"official":{"repos":["JunjieHu/ReCo-RL"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-corpus-free-state2seq-user-simulator-for","slug":"a-corpus-free-state2seq-user-simulator-for","title":"A Corpus-free State2Seq User Simulator for Task-oriented Dialogue","date":"2019-09-10","arxiv_id":"1909.04448","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-reproducibility-by-evaluating","slug":"a-survey-on-reproducibility-by-evaluating","title":"A Survey on Reproducibility by Evaluating Deep Reinforcement Learning Algorithms on Real-World Robots","date":"2019-09-09","arxiv_id":"1909.03772","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-survey-on-reproducibility-by-evaluating#ran","syntology_url":"https://syntology.ai/paper/1909.03772","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.03772"}},"official":{"repos":["dti-research/SenseActExperiments"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ac-teach-a-bayesian-actor-critic-method-for","slug":"ac-teach-a-bayesian-actor-critic-method-for","title":"AC-Teach: A Bayesian Actor-Critic Method for Policy Learning with an Ensemble of Suboptimal Teachers","date":"2019-09-09","arxiv_id":"1909.04121","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/ac-teach-a-bayesian-actor-critic-method-for#ran","syntology_url":"https://syntology.ai/paper/1909.04121","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.04121"}},"official":null}},{"url":"/paper/adversarial-policy-gradient-for-deep-learning","slug":"adversarial-policy-gradient-for-deep-learning","title":"Adversarial Policy Gradient for Deep Learning Image Augmentation","date":"2019-09-09","arxiv_id":"1909.04108","repositories_listed":1,"syntology":null},{"url":"/paper/clickbait-sensational-headline-generation","slug":"clickbait-sensational-headline-generation","title":"Clickbait? Sensational Headline Generation with Auto-tuned Reinforcement Learning","date":"2019-09-09","arxiv_id":"1909.03582","repositories_listed":1,"syntology":null},{"url":"/paper/regularized-anderson-acceleration-for-off","slug":"regularized-anderson-acceleration-for-off","title":"Regularized Anderson Acceleration for Off-Policy Deep Reinforcement Learning","date":"2019-09-07","arxiv_id":"1909.03245","repositories_listed":1,"syntology":null},{"url":"/paper/drlviz-understanding-decisions-and-memory-in","slug":"drlviz-understanding-decisions-and-memory-in","title":"DRLViz: Understanding Decisions and Memory in Deep Reinforcement Learning","date":"2019-09-06","arxiv_id":"1909.02982","repositories_listed":1,"syntology":null},{"url":"/paper/learning-action-transferable-policy-with","slug":"learning-action-transferable-policy-with","title":"Learning Action-Transferable Policy with Action Embedding","date":"2019-09-05","arxiv_id":"1909.02291","repositories_listed":1,"syntology":null},{"url":"/paper/rewarding-coreference-resolvers-for-being","slug":"rewarding-coreference-resolvers-for-being","title":"Rewarding Coreference Resolvers for Being Consistent with World Knowledge","date":"2019-09-05","arxiv_id":"1909.02392","repositories_listed":1,"syntology":null},{"url":"/paper/spatiotemporally-constrained-action-space","slug":"spatiotemporally-constrained-action-space","title":"Spatiotemporally Constrained Action Space Attacks on Deep Reinforcement Learning Agents","date":"2019-09-05","arxiv_id":"1909.02583","repositories_listed":1,"syntology":null},{"url":"/paper/no-press-diplomacy-modeling-multi-agent","slug":"no-press-diplomacy-modeling-multi-agent","title":"No Press Diplomacy: Modeling Multi-Agent Gameplay","date":"2019-09-04","arxiv_id":"1909.02128","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/no-press-diplomacy-modeling-multi-agent#ran","syntology_url":"https://syntology.ai/paper/1909.02128","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.02128"}},"official":{"repos":["diplomacy/research"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-to-build-user-simulators-to-train-rl","slug":"how-to-build-user-simulators-to-train-rl","title":"How to Build User Simulators to Train RL-based Dialog Systems","date":"2019-09-03","arxiv_id":"1909.01388","repositories_listed":1,"syntology":null},{"url":"/paper/an-open-source-framework-for-adaptive-traffic","slug":"an-open-source-framework-for-adaptive-traffic","title":"An Open-Source Framework for Adaptive Traffic Signal Control","date":"2019-09-01","arxiv_id":"1909.00395","repositories_listed":1,"syntology":null},{"url":"/paper/generating-classical-chinese-poems-from","slug":"generating-classical-chinese-poems-from","title":"Generating Classical Chinese Poems from Vernacular Chinese","date":"2019-08-31","arxiv_id":"1909.00279","repositories_listed":1,"syntology":null},{"url":"/paper/meta-learning-with-warped-gradient-descent","slug":"meta-learning-with-warped-gradient-descent","title":"Meta-Learning with Warped Gradient Descent","date":"2019-08-30","arxiv_id":"1909.00025","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/meta-learning-with-warped-gradient-descent#ran","syntology_url":"https://syntology.ai/paper/1909.00025","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.00025"}},"official":{"repos":["flennerhag/warpgrad"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/an-empirical-comparison-on-imitation-learning","slug":"an-empirical-comparison-on-imitation-learning","title":"An Empirical Comparison on Imitation Learning and Reinforcement Learning for Paraphrase Generation","date":"2019-08-28","arxiv_id":"1908.10835","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/an-empirical-comparison-on-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/1908.10835","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.10835"}},"official":{"repos":["ddddwy/Reinforce-Paraphrase-Generation"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/guided-dialog-policy-learning-reward","slug":"guided-dialog-policy-learning-reward","title":"Guided Dialog Policy Learning: Reward Estimation for Multi-Domain Task-Oriented Dialog","date":"2019-08-28","arxiv_id":"1908.10719","repositories_listed":1,"syntology":null},{"url":"/paper/intelligent-active-queue-management-using","slug":"intelligent-active-queue-management-using","title":"Intelligent Active Queue Management Using Explicit Congestion Notification","date":"2019-08-28","arxiv_id":"1909.08386","repositories_listed":1,"syntology":null},{"url":"/paper/interactive-language-learning-by-question","slug":"interactive-language-learning-by-question","title":"Interactive Language Learning by Question Answering","date":"2019-08-28","arxiv_id":"1908.10909","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/interactive-language-learning-by-question#ran","syntology_url":"https://syntology.ai/paper/1908.10909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.10909"}},"official":{"repos":["xingdi-eric-yuan/qait_public"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/continuous-value-iteration-cvi-reinforcement","slug":"continuous-value-iteration-cvi-reinforcement","title":"Continuous Value Iteration (CVI) Reinforcement Learning and Imaginary Experience Replay (IER) for learning multi-goal, continuous action and state space controllers","date":"2019-08-27","arxiv_id":"1908.10255","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-text-classification-with","slug":"hierarchical-text-classification-with","title":"Hierarchical Text Classification with Reinforced Label Assignment","date":"2019-08-27","arxiv_id":"1908.10419","repositories_listed":1,"syntology":null},{"url":"/paper/universal-policies-to-learn-them-all","slug":"universal-policies-to-learn-them-all","title":"Universal Policies to Learn Them All","date":"2019-08-24","arxiv_id":"1908.09184","repositories_listed":1,"syntology":null},{"url":"/paper/double-reinforcement-learning-for-efficient","slug":"double-reinforcement-learning-for-efficient","title":"Double Reinforcement Learning for Efficient Off-Policy Evaluation in Markov Decision Processes","date":"2019-08-22","arxiv_id":"1908.08526","repositories_listed":1,"syntology":null},{"url":"/paper/opponent-aware-reinforcement-learning","slug":"opponent-aware-reinforcement-learning","title":"Opponent Aware Reinforcement Learning","date":"2019-08-22","arxiv_id":"1908.08773","repositories_listed":1,"syntology":null},{"url":"/paper/automated-quantum-programming-via","slug":"automated-quantum-programming-via","title":"Automated quantum programming via reinforcement learning for combinatorial optimization","date":"2019-08-21","arxiv_id":"1908.08054","repositories_listed":1,"syntology":null},{"url":"/paper/araml-a-stable-adversarial-training-framework","slug":"araml-a-stable-adversarial-training-framework","title":"ARAML: A Stable Adversarial Training Framework for Text Generation","date":"2019-08-20","arxiv_id":"1908.07195","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 4 unverified","sample_list":"/paper/araml-a-stable-adversarial-training-framework#ran","syntology_url":"https://syntology.ai/paper/1908.07195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.07195"}},"official":{"repos":["kepei1106/ARAML"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"url":"/paper/an-autonomous-performance-testing-framework","slug":"an-autonomous-performance-testing-framework","title":"An Autonomous Performance Testing Framework using Self-Adaptive Fuzzy Reinforcement Learning","date":"2019-08-19","arxiv_id":"1908.06900","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-in-world-earth","slug":"deep-reinforcement-learning-in-world-earth","title":"Deep reinforcement learning in World-Earth system models to discover sustainable management strategies","date":"2019-08-15","arxiv_id":"1908.05567","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/deep-reinforcement-learning-in-world-earth#ran","syntology_url":"https://syntology.ai/paper/1908.05567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.05567"}},"official":{"repos":["fstrnad/pyDRLinWESM"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/imitation-learning-for-sentence-generation","slug":"imitation-learning-for-sentence-generation","title":"Imitation Learning for Sentence Generation with Dilated Convolutions Using Adversarial Training","date":"2019-08-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-graph-to","slug":"reinforcement-learning-based-graph-to","title":"Reinforcement Learning Based Graph-to-Sequence Model for Natural Question Generation","date":"2019-08-14","arxiv_id":"1908.04942","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/reinforcement-learning-based-graph-to#ran","syntology_url":"https://syntology.ai/paper/1908.04942","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.04942"}},"official":{"repos":["hugochan/RL-based-Graph2Seq-for-NQG"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-diverse-and-accurate-image-captions","slug":"towards-diverse-and-accurate-image-captions","title":"Towards Diverse and Accurate Image Captions via Reinforcing Determinantal Point Process","date":"2019-08-14","arxiv_id":"1908.04919","repositories_listed":1,"syntology":null},{"url":"/paper/videonavqa-bridging-the-gap-between-visual","slug":"videonavqa-bridging-the-gap-between-visual","title":"VideoNavQA: Bridging the Gap between Visual and Embodied Question Answering","date":"2019-08-14","arxiv_id":"1908.04950","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/videonavqa-bridging-the-gap-between-visual#ran","syntology_url":"https://syntology.ai/paper/1908.04950","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.04950"}},"official":{"repos":["catalina17/VideoNavQA"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/generative-question-refinement-with-deep","slug":"generative-question-refinement-with-deep","title":"Generative Question Refinement with Deep Reinforcement Learning in Retrieval-based QA System","date":"2019-08-13","arxiv_id":"1908.05604","repositories_listed":1,"syntology":null},{"url":"/paper/is-deep-reinforcement-learning-really","slug":"is-deep-reinforcement-learning-really","title":"Is Deep Reinforcement Learning Really Superhuman on Atari? Leveling the playing field","date":"2019-08-13","arxiv_id":"1908.04683","repositories_listed":1,"syntology":null},{"url":"/paper/a-review-on-deep-reinforcement-learning-for","slug":"a-review-on-deep-reinforcement-learning-for","title":"A review on Deep Reinforcement Learning for Fluid Mechanics","date":"2019-08-12","arxiv_id":"1908.04127","repositories_listed":1,"syntology":null},{"url":"/paper/vision-based-navigation-using-deep","slug":"vision-based-navigation-using-deep","title":"Vision-based Navigation Using Deep Reinforcement Learning","date":"2019-08-08","arxiv_id":"1908.03627","repositories_listed":1,"syntology":null}],"record_sha256":"81ca47eb0ca992173450f447062b89e37da844e1db8c580da4ab2e9c89484cf2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}