{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/q-learning/papers/4","list_of":"/task/q-learning","task":"Q-Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":20,"rows_per_page":100,"rows":[301,400],"of":1918,"counts":{"archive_papers_tagged":1918,"with_a_code_link":463,"where_syntology_ran_a_sample":119,"not_listed_spam_title":0,"listed":1918,"listed_where_code_ran":119,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":102,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":102,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/q-learning","prev":"/task/q-learning/papers/3","next":"/task/q-learning/papers/5","papers":[{"url":"/paper/catastrophic-interference-in-reinforcement","slug":"catastrophic-interference-in-reinforcement","title":"Catastrophic Interference in Reinforcement Learning: A Solution Based on Context Division and Knowledge Distillation","date":"2021-09-01","arxiv_id":"2109.00525","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/catastrophic-interference-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2109.00525","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.00525"}},"official":{"repos":["sweety-dm/interference-aware-deep-q-learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/saber-data-driven-motion-planner-for","slug":"saber-data-driven-motion-planner-for","title":"SABER: Data-Driven Motion Planner for Autonomously Navigating Heterogeneous Robots","date":"2021-08-03","arxiv_id":"2108.01262","repositories_listed":1,"syntology":null},{"url":"/paper/a-dqn-based-approach-to-finding-precise","slug":"a-dqn-based-approach-to-finding-precise","title":"A DQN-based Approach to Finding Precise Evidences for Fact Verification","date":"2021-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/backprop-free-reinforcement-learning-with","slug":"backprop-free-reinforcement-learning-with","title":"Backprop-Free Reinforcement Learning with Active Neural Generative Coding","date":"2021-07-10","arxiv_id":"2107.07046","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/backprop-free-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2107.07046","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.07046"}},"official":{"repos":["ago109/active-neural-generative-coding"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/computational-benefits-of-intermediate","slug":"computational-benefits-of-intermediate","title":"Computational Benefits of Intermediate Rewards for Goal-Reaching Policy Learning","date":"2021-07-08","arxiv_id":"2107.03961","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/computational-benefits-of-intermediate#ran","syntology_url":"https://syntology.ai/paper/2107.03961","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.03961"}},"official":{"repos":["kebaek/minigrid"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ensemble-and-auxiliary-tasks-for-data","slug":"ensemble-and-auxiliary-tasks-for-data","title":"Ensemble and Auxiliary Tasks for Data-Efficient Deep Reinforcement Learning","date":"2021-07-05","arxiv_id":"2107.01904","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ensemble-and-auxiliary-tasks-for-data#ran","syntology_url":"https://syntology.ai/paper/2107.01904","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.01904"}},"official":{"repos":["NUS-LID/RENAULT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/distilling-reinforcement-learning-tricks-for","slug":"distilling-reinforcement-learning-tricks-for","title":"Distilling Reinforcement Learning Tricks for Video Games","date":"2021-07-01","arxiv_id":"2107.00703","repositories_listed":1,"syntology":null},{"url":"/paper/towards-self-organized-control-using-neural","slug":"towards-self-organized-control-using-neural","title":"Towards self-organized control: Using neural cellular automata to robustly control a cart-pole agent","date":"2021-06-29","arxiv_id":"2106.15240","repositories_listed":1,"syntology":null},{"url":"/paper/coarse-to-fine-q-attention-efficient-learning","slug":"coarse-to-fine-q-attention-efficient-learning","title":"Coarse-to-Fine Q-attention: Efficient Learning for Visual Robotic Manipulation via Discretisation","date":"2021-06-23","arxiv_id":"2106.12534","repositories_listed":1,"syntology":null},{"url":"/paper/q-learning-lagrange-policies-for-multi-action","slug":"q-learning-lagrange-policies-for-multi-action","title":"Q-Learning Lagrange Policies for Multi-Action Restless Bandits","date":"2021-06-22","arxiv_id":"2106.12024","repositories_listed":1,"syntology":{"n":17,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/q-learning-lagrange-policies-for-multi-action#ran","syntology_url":"https://syntology.ai/paper/2106.12024","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.12024"}},"official":{"repos":["killian-34/MAIQL_and_LPQL"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-phy-layer","slug":"reinforcement-learning-for-phy-layer","title":"Reinforcement Learning for Physical Layer Communications","date":"2021-06-22","arxiv_id":"2106.11595","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-heuristic-multi-agent-path","slug":"distributed-heuristic-multi-agent-path","title":"Distributed Heuristic Multi-Agent Path Finding with Communication","date":"2021-06-21","arxiv_id":"2106.11365","repositories_listed":1,"syntology":null},{"url":"/paper/text-generation-with-efficient-soft-q","slug":"text-generation-with-efficient-soft-q","title":"Efficient (Soft) Q-Learning for Text Generation with Limited Good Data","date":"2021-06-14","arxiv_id":"2106.07704","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/text-generation-with-efficient-soft-q#ran","syntology_url":"https://syntology.ai/paper/2106.07704","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.07704"}},"official":{"repos":["HanGuo97/soft-Q-learning-for-text-generation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/believe-what-you-see-implicit-constraint","slug":"believe-what-you-see-implicit-constraint","title":"Believe What You See: Implicit Constraint Approach for Offline Multi-Agent Reinforcement Learning","date":"2021-06-07","arxiv_id":"2106.03400","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-target-networks-improving-deep-q","slug":"beyond-target-networks-improving-deep-q","title":"Bridging the Gap Between Target Networks and Functional Regularization","date":"2021-06-04","arxiv_id":"2106.02613","repositories_listed":1,"syntology":null},{"url":"/paper/shaq-incorporating-shapley-value-theory-into","slug":"shaq-incorporating-shapley-value-theory-into","title":"SHAQ: Incorporating Shapley Value Theory into Multi-Agent Q-Learning","date":"2021-05-31","arxiv_id":"2105.15013","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/shaq-incorporating-shapley-value-theory-into#ran","syntology_url":"https://syntology.ai/paper/2105.15013","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.15013"}},"official":{"repos":["hsvgbkhgbv/shapley-q-learning"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/a-comparison-of-reward-functions-in-q","slug":"a-comparison-of-reward-functions-in-q","title":"A Comparison of Reward Functions in Q-Learning Applied to a Cart Position Problem","date":"2021-05-25","arxiv_id":"2105.11617","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-optimal-3","slug":"deep-reinforcement-learning-for-optimal-3","title":"Deep Reinforcement Learning for Optimal Stopping with Application in Financial Engineering","date":"2021-05-19","arxiv_id":"2105.08877","repositories_listed":1,"syntology":null},{"url":"/paper/hasco-towards-agile-hardware-and-software-co","slug":"hasco-towards-agile-hardware-and-software-co","title":"HASCO: Towards Agile HArdware and Software CO-design for Tensor Computation","date":"2021-05-04","arxiv_id":"2105.01585","repositories_listed":1,"syntology":null},{"url":"/paper/action-candidate-based-clipped-double-q","slug":"action-candidate-based-clipped-double-q","title":"Action Candidate Based Clipped Double Q-learning for Discrete and Continuous Action Tasks","date":"2021-05-03","arxiv_id":"2105.00704","repositories_listed":1,"syntology":null},{"url":"/paper/robotic-surgery-with-lean-reinforcement","slug":"robotic-surgery-with-lean-reinforcement","title":"Robotic Surgery With Lean Reinforcement Learning","date":"2021-05-03","arxiv_id":"2105.01006","repositories_listed":1,"syntology":null},{"url":"/paper/low-rank-state-action-value-function","slug":"low-rank-state-action-value-function","title":"Low-rank State-action Value-function Approximation","date":"2021-04-18","arxiv_id":"2104.08805","repositories_listed":1,"syntology":null},{"url":"/paper/group-equivariant-neural-architecture-search","slug":"group-equivariant-neural-architecture-search","title":"Autoequivariant Network Search via Group Decomposition","date":"2021-04-10","arxiv_id":"2104.04848","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-market-making-by-reinforcement","slug":"optimal-market-making-by-reinforcement","title":"Optimal Market Making by Reinforcement Learning","date":"2021-04-08","arxiv_id":"2104.04036","repositories_listed":1,"syntology":null},{"url":"/paper/ucb-momentum-q-learning-correcting-the-bias","slug":"ucb-momentum-q-learning-correcting-the-bias","title":"UCB Momentum Q-learning: Correcting the bias without forgetting","date":"2021-03-01","arxiv_id":"2103.01312","repositories_listed":1,"syntology":null},{"url":"/paper/balancing-rational-and-other-regarding","slug":"balancing-rational-and-other-regarding","title":"Balancing Rational and Other-Regarding Preferences in Cooperative-Competitive Environments","date":"2021-02-24","arxiv_id":"2102.12307","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-algorithmic-collusion-with","slug":"understanding-algorithmic-collusion-with","title":"Understanding algorithmic collusion with experience replay","date":"2021-02-18","arxiv_id":"2102.09139","repositories_listed":1,"syntology":null},{"url":"/paper/dfac-framework-factorizing-the-value-function","slug":"dfac-framework-factorizing-the-value-function","title":"DFAC Framework: Factorizing the Value Function via Quantile Mixture for Multi-Agent Distributional Q-Learning","date":"2021-02-16","arxiv_id":"2102.07936","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/dfac-framework-factorizing-the-value-function#ran","syntology_url":"https://syntology.ai/paper/2102.07936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.07936"}},"official":{"repos":["j3soon/dfac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-deep-graph-generative-models-for","slug":"benchmarking-deep-graph-generative-models-for","title":"Benchmarking Deep Graph Generative Models for Optimizing New Drug Molecules for COVID-19","date":"2021-02-09","arxiv_id":"2102.04977","repositories_listed":1,"syntology":null},{"url":"/paper/variation-resistant-q-learning-controlling","slug":"variation-resistant-q-learning-controlling","title":"Variation-resistant Q-learning: Controlling and Utilizing Estimation Bias in Reinforcement Learning for Better Performance","date":"2021-02-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/breaking-the-deadly-triad-with-a-target","slug":"breaking-the-deadly-triad-with-a-target","title":"Breaking the Deadly Triad with a Target Network","date":"2021-01-21","arxiv_id":"2101.08862","repositories_listed":1,"syntology":null},{"url":"/paper/continuous-deep-q-learning-with-simulator-for","slug":"continuous-deep-q-learning-with-simulator-for","title":"Continuous Deep Q-Learning with Simulator for Stabilization of Uncertain Discrete-Time Systems","date":"2021-01-13","arxiv_id":"2101.05640","repositories_listed":1,"syntology":null},{"url":"/paper/simulating-sql-injection-vulnerability","slug":"simulating-sql-injection-vulnerability","title":"Simulating SQL Injection Vulnerability Exploitation Using Q-Learning Reinforcement Learning Agents","date":"2021-01-08","arxiv_id":"2101.03118","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-trust-region-learning","slug":"multi-agent-trust-region-learning","title":"Multi-Agent Trust Region Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/popo-pessimistic-offline-policy-optimization","slug":"popo-pessimistic-offline-policy-optimization","title":"POPO: Pessimistic Offline Policy Optimization","date":"2020-12-26","arxiv_id":"2012.13682","repositories_listed":1,"syntology":null},{"url":"/paper/model-free-and-bayesian-ensembling-model","slug":"model-free-and-bayesian-ensembling-model","title":"Model-free and Bayesian Ensembling Model-based Deep Reinforcement Learning for Particle Accelerator Control Demonstrated on the FERMI FEL","date":"2020-12-17","arxiv_id":"2012.09737","repositories_listed":1,"syntology":null},{"url":"/paper/combining-reinforcement-learning-with-lin","slug":"combining-reinforcement-learning-with-lin","title":"Combining Reinforcement Learning with Lin-Kernighan-Helsgaun Algorithm for the Traveling Salesman Problem","date":"2020-12-08","arxiv_id":"2012.04461","repositories_listed":1,"syntology":null},{"url":"/paper/can-q-learning-with-graph-networks-learn-a","slug":"can-q-learning-with-graph-networks-learn-a","title":"Can Q-Learning with Graph Networks Learn a Generalizable Branching Heuristic for a SAT Solver?","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-principle-of-least-action-with","slug":"learning-principle-of-least-action-with","title":"Learning Principle of Least Action with Reinforcement Learning","date":"2020-11-24","arxiv_id":"2011.11891","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-contention-window-design-using-deep","slug":"adaptive-contention-window-design-using-deep","title":"Adaptive Contention Window Design using Deep Q-learning","date":"2020-11-18","arxiv_id":"2011.09418","repositories_listed":1,"syntology":null},{"url":"/paper/control-with-adaptive-q-learning","slug":"control-with-adaptive-q-learning","title":"Control with adaptive Q-learning","date":"2020-11-03","arxiv_id":"2011.02141","repositories_listed":1,"syntology":null},{"url":"/paper/deep-jump-q-evaluation-for-offline-policy-1","slug":"deep-jump-q-evaluation-for-offline-policy-1","title":"Deep Jump Learning for Off-Policy Evaluation in Continuous Treatment Settings","date":"2020-10-29","arxiv_id":"2010.15963","repositories_listed":1,"syntology":null},{"url":"/paper/hamilton-jacobi-deep-q-learning-for","slug":"hamilton-jacobi-deep-q-learning-for","title":"Hamilton-Jacobi Deep Q-Learning for Deterministic Continuous-Time Systems with Lipschitz Continuous Controls","date":"2020-10-27","arxiv_id":"2010.14087","repositories_listed":1,"syntology":null},{"url":"/paper/instance-weighted-incremental-evolution","slug":"instance-weighted-incremental-evolution","title":"Instance Weighted Incremental Evolution Strategies for Reinforcement Learning in Dynamic Environments","date":"2020-10-09","arxiv_id":"2010.04605","repositories_listed":1,"syntology":null},{"url":"/paper/q-learning-with-language-model-for-edit-based","slug":"q-learning-with-language-model-for-edit-based","title":"Q-learning with Language Model for Edit-based Unsupervised Summarization","date":"2020-10-09","arxiv_id":"2010.04379","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/q-learning-with-language-model-for-edit-based#ran","syntology_url":"https://syntology.ai/paper/2010.04379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.04379"}},"official":{"repos":["kohilin/ealm"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/bootstrapped-q-learning-with-context-relevant","slug":"bootstrapped-q-learning-with-context-relevant","title":"Bootstrapped Q-learning with Context Relevant Observation Pruning to Generalize in Text-based Games","date":"2020-09-24","arxiv_id":"2009.11896","repositories_listed":1,"syntology":null},{"url":"/paper/energy-based-surprise-minimization-for-multi","slug":"energy-based-surprise-minimization-for-multi","title":"Energy-based Surprise Minimization for Multi-Agent Value Factorization","date":"2020-09-16","arxiv_id":"2009.09842","repositories_listed":1,"syntology":null},{"url":"/paper/deep-active-inference-for-partially","slug":"deep-active-inference-for-partially","title":"Deep Active Inference for Partially Observable MDPs","date":"2020-09-08","arxiv_id":"2009.03622","repositories_listed":1,"syntology":null},{"url":"/paper/table2charts-learning-shared-representations","slug":"table2charts-learning-shared-representations","title":"Table2Charts: Recommending Charts by Learning Shared Table Representations","date":"2020-08-24","arxiv_id":"2008.11015","repositories_listed":1,"syntology":null},{"url":"/paper/pc-pg-policy-cover-directed-exploration-for","slug":"pc-pg-policy-cover-directed-exploration-for","title":"PC-PG: Policy Cover Directed Exploration for Provable Policy Gradient Learning","date":"2020-07-16","arxiv_id":"2007.08459","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/pc-pg-policy-cover-directed-exploration-for#ran","syntology_url":"https://syntology.ai/paper/2007.08459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.08459"}},"official":null}},{"url":"/paper/single-partition-adaptive-q-learning","slug":"single-partition-adaptive-q-learning","title":"Single-partition adaptive Q-learning","date":"2020-07-14","arxiv_id":"2007.06741","repositories_listed":1,"syntology":null},{"url":"/paper/provably-efficient-double-q-learning","slug":"provably-efficient-double-q-learning","title":"The Mean-Squared Error of Double Q-Learning","date":"2020-07-09","arxiv_id":"2007.05034","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/provably-efficient-double-q-learning#ran","syntology_url":"https://syntology.ai/paper/2007.05034","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.05034"}},"official":{"repos":["wentaoweng/The-Mean-Squared-Error-of-Double-Q-Learning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sunrise-a-simple-unified-framework-for","slug":"sunrise-a-simple-unified-framework-for","title":"SUNRISE: A Simple Unified Framework for Ensemble Learning in Deep Reinforcement Learning","date":"2020-07-09","arxiv_id":"2007.04938","repositories_listed":1,"syntology":null},{"url":"/paper/neural-interactive-collaborative-filtering","slug":"neural-interactive-collaborative-filtering","title":"Neural Interactive Collaborative Filtering","date":"2020-07-04","arxiv_id":"2007.02095","repositories_listed":1,"syntology":null},{"url":"/paper/gradient-temporal-difference-learning-with","slug":"gradient-temporal-difference-learning-with","title":"Gradient Temporal-Difference Learning with Regularized Corrections","date":"2020-07-01","arxiv_id":"2007.00611","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gradient-temporal-difference-learning-with#ran","syntology_url":"https://syntology.ai/paper/2007.00611","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.00611"}},"official":{"repos":["rlai-lab/Regularized-GradientTD"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/group-equivariant-deep-reinforcement-learning","slug":"group-equivariant-deep-reinforcement-learning","title":"Group Equivariant Deep Reinforcement Learning","date":"2020-07-01","arxiv_id":"2007.03437","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/group-equivariant-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2007.03437","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.03437"}},"official":{"repos":["arnab39/EquivariantDQN"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/image-classification-by-reinforcement","slug":"image-classification-by-reinforcement","title":"Image Classification by Reinforcement Learning with Two-State Q-Learning","date":"2020-06-28","arxiv_id":"2007.01298","repositories_listed":1,"syntology":null},{"url":"/paper/lookahead-bounded-q-learning","slug":"lookahead-bounded-q-learning","title":"Lookahead-Bounded Q-Learning","date":"2020-06-28","arxiv_id":"2006.15690","repositories_listed":1,"syntology":null},{"url":"/paper/overfitting-and-optimization-in-offline","slug":"overfitting-and-optimization-in-offline","title":"Offline Contextual Bandits with Overparameterized Models","date":"2020-06-27","arxiv_id":"2006.15368","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-visual-navigation-by-watching","slug":"semantic-visual-navigation-by-watching","title":"Semantic Visual Navigation by Watching YouTube Videos","date":"2020-06-17","arxiv_id":"2006.10034","repositories_listed":1,"syntology":null},{"url":"/paper/a-novel-update-mechanism-for-q-networks-based","slug":"a-novel-update-mechanism-for-q-networks-based","title":"A Novel Update Mechanism for Q-Networks Based On Extreme Learning Machines","date":"2020-06-04","arxiv_id":"2006.02986","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-determinantal-q-learning","slug":"multi-agent-determinantal-q-learning","title":"Multi-Agent Determinantal Q-Learning","date":"2020-06-02","arxiv_id":"2006.01482","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/multi-agent-determinantal-q-learning#ran","syntology_url":"https://syntology.ai/paper/2006.01482","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.01482"}},"official":{"repos":["QDPP-GitHub/QDPP"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/modeling-penetration-testing-with","slug":"modeling-penetration-testing-with","title":"Modeling Penetration Testing with Reinforcement Learning Using Capture-the-Flag Challenges: Trade-offs between Model-free Learning and A Priori Knowledge","date":"2020-05-26","arxiv_id":"2005.12632","repositories_listed":1,"syntology":null},{"url":"/paper/spatial-action-maps-for-mobile-manipulation","slug":"spatial-action-maps-for-mobile-manipulation","title":"Spatial Action Maps for Mobile Manipulation","date":"2020-04-20","arxiv_id":"2004.09141","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spatial-action-maps-for-mobile-manipulation#ran","syntology_url":"https://syntology.ai/paper/2004.09141","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.09141"}},"official":{"repos":["jimmyyhwu/spatial-action-maps"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-punishment-and-reward-backfill-for-deep","slug":"self-punishment-and-reward-backfill-for-deep","title":"Self Punishment and Reward Backfill for Deep Q-Learning","date":"2020-04-10","arxiv_id":"2004.05002","repositories_listed":1,"syntology":null},{"url":"/paper/augmented-q-imitation-learning-aqil","slug":"augmented-q-imitation-learning-aqil","title":"Augmented Q Imitation Learning (AQIL)","date":"2020-03-31","arxiv_id":"2004.00993","repositories_listed":1,"syntology":null},{"url":"/paper/using-deep-reinforcement-learning-methods-for","slug":"using-deep-reinforcement-learning-methods-for","title":"Using Deep Reinforcement Learning Methods for Autonomous Vessels in 2D Environments","date":"2020-03-23","arxiv_id":"2003.10249","repositories_listed":1,"syntology":null},{"url":"/paper/conqur-mitigating-delusional-bias-in-deep-q-1","slug":"conqur-mitigating-delusional-bias-in-deep-q-1","title":"ConQUR: Mitigating Delusional Bias in Deep Q-learning","date":"2020-02-27","arxiv_id":"2002.12399","repositories_listed":1,"syntology":null},{"url":"/paper/optimistic-exploration-even-with-a-1","slug":"optimistic-exploration-even-with-a-1","title":"Optimistic Exploration even with a Pessimistic Initialisation","date":"2020-02-26","arxiv_id":"2002.12174","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimistic-exploration-even-with-a-1#ran","syntology_url":"https://syntology.ai/paper/2002.12174","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.12174"}},"official":{"repos":["oxwhirl/opiq"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/maxmin-q-learning-controlling-the-estimation-1","slug":"maxmin-q-learning-controlling-the-estimation-1","title":"Maxmin Q-learning: Controlling the Estimation Bias of Q-learning","date":"2020-02-16","arxiv_id":"2002.06487","repositories_listed":1,"syntology":null},{"url":"/paper/a-stochastic-game-framework-for-efficient","slug":"a-stochastic-game-framework-for-efficient","title":"A Stochastic Game Framework for Efficient Energy Management in Microgrid Networks","date":"2020-02-06","arxiv_id":"2002.02084","repositories_listed":1,"syntology":null},{"url":"/paper/discriminator-soft-actor-critic-without","slug":"discriminator-soft-actor-critic-without","title":"Discriminator Soft Actor Critic without Extrinsic Rewards","date":"2020-01-19","arxiv_id":"2001.06808","repositories_listed":1,"syntology":null},{"url":"/paper/an-optimistic-perspective-on-offline-deep","slug":"an-optimistic-perspective-on-offline-deep","title":"An Optimistic Perspective on Offline Deep Reinforcement Learning","date":"2020-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-an-interpretable-traffic-signal","slug":"learning-an-interpretable-traffic-signal","title":"Learning an Interpretable Traffic Signal Control Policy","date":"2019-12-23","arxiv_id":"1912.11023","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-cooperative-multi-agent","slug":"hierarchical-cooperative-multi-agent","title":"Hierarchical Cooperative Multi-Agent Reinforcement Learning with Skill Discovery","date":"2019-12-07","arxiv_id":"1912.03558","repositories_listed":1,"syntology":null},{"url":"/paper/privacy-preserving-q-learning-with-functional","slug":"privacy-preserving-q-learning-with-functional","title":"Privacy-Preserving Q-Learning with Functional Noise in Continuous Spaces","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/propagating-uncertainty-in-reinforcement","slug":"propagating-uncertainty-in-reinforcement","title":"Propagating Uncertainty in Reinforcement Learning via Wasserstein Barycenters","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/qmr-q-learning-based-multi-objective","slug":"qmr-q-learning-based-multi-objective","title":"QMR:Q-learning based Multi-objective optimization Routing protocol for Flying Ad Hoc Networks","date":"2019-11-27","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/join-query-optimization-with-deep","slug":"join-query-optimization-with-deep","title":"Join Query Optimization with Deep Reinforcement Learning Algorithms","date":"2019-11-26","arxiv_id":"1911.11689","repositories_listed":1,"syntology":null},{"url":"/paper/on-solving-the-2-dimensional-greedy-shooter","slug":"on-solving-the-2-dimensional-greedy-shooter","title":"On Solving the 2-Dimensional Greedy Shooter Problem for UAVs","date":"2019-11-02","arxiv_id":"1911.01419","repositories_listed":1,"syntology":null},{"url":"/paper/generalized-speedy-q-learning","slug":"generalized-speedy-q-learning","title":"Generalized Speedy Q-learning","date":"2019-11-01","arxiv_id":"1911.00397","repositories_listed":1,"syntology":null},{"url":"/paper/bail-best-action-imitation-learning-for-batch-1","slug":"bail-best-action-imitation-learning-for-batch-1","title":"BAIL: Best-Action Imitation Learning for Batch Deep Reinforcement Learning","date":"2019-10-27","arxiv_id":"1910.12179","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bail-best-action-imitation-learning-for-batch-1#ran","syntology_url":"https://syntology.ai/paper/1910.12179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.12179"}},"official":{"repos":["lanyavik/BAIL"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/momentum-based-accelerated-q-learning","slug":"momentum-based-accelerated-q-learning","title":"Momentum-based Accelerated Q-learning","date":"2019-10-23","arxiv_id":"1910.11673","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-data-augmentation-by-learning-the","slug":"automatic-data-augmentation-by-learning-the","title":"Automatic Data Augmentation by Learning the Deterministic Policy","date":"2019-10-18","arxiv_id":"1910.08343","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-discretization-for-episodic","slug":"adaptive-discretization-for-episodic","title":"Adaptive Discretization for Episodic Reinforcement Learning in Metric Spaces","date":"2019-10-17","arxiv_id":"1910.08151","repositories_listed":1,"syntology":null},{"url":"/paper/combining-no-regret-and-q-learning","slug":"combining-no-regret-and-q-learning","title":"Combining No-regret and Q-learning","date":"2019-10-07","arxiv_id":"1910.03094","repositories_listed":1,"syntology":null},{"url":"/paper/a-simulation-of-uav-power-optimization-via","slug":"a-simulation-of-uav-power-optimization-via","title":"Visual Exploration and Energy-aware Path Planning via Reinforcement Learning","date":"2019-09-26","arxiv_id":"1909.12217","repositories_listed":1,"syntology":null},{"url":"/paper/demystifying-active-inference","slug":"demystifying-active-inference","title":"Active inference: demystified and compared","date":"2019-09-24","arxiv_id":"1909.10863","repositories_listed":1,"syntology":null},{"url":"/paper/modelicagym-applying-reinforcement-learning","slug":"modelicagym-applying-reinforcement-learning","title":"ModelicaGym: Applying Reinforcement Learning to Modelica Models","date":"2019-09-18","arxiv_id":"1909.08604","repositories_listed":1,"syntology":null},{"url":"/paper/isl-optimal-policy-learning-with-optimal","slug":"isl-optimal-policy-learning-with-optimal","title":"ISL: A novel approach for deep exploration","date":"2019-09-13","arxiv_id":"1909.06293","repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-learning-approach-to-grasping-the","slug":"a-deep-learning-approach-to-grasping-the","title":"A Deep Learning Approach to Grasping the Invisible","date":"2019-09-11","arxiv_id":"1909.04840","repositories_listed":1,"syntology":null},{"url":"/paper/learn-how-to-cook-a-new-recipe-in-a-new-house","slug":"learn-how-to-cook-a-new-recipe-in-a-new-house","title":"Learn How to Cook a New Recipe in a New House: Using Map Familiarization, Curriculum Learning, and Bandit Feedback to Learn Families of Text-Based Adventure Games","date":"2019-08-13","arxiv_id":"1908.04777","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learn-how-to-cook-a-new-recipe-in-a-new-house#ran","syntology_url":"https://syntology.ai/paper/1908.04777","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.04777"}},"official":{"repos":["yinxusen/deepword"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/control-of-nonlinear-complex-and-black-boxed","slug":"control-of-nonlinear-complex-and-black-boxed","title":"Control of nonlinear, complex and black-boxed greenhouse system with reinforcement learning","date":"2019-07-30","arxiv_id":"1907.12690","repositories_listed":1,"syntology":null},{"url":"/paper/towards-model-based-reinforcement-learning","slug":"towards-model-based-reinforcement-learning","title":"Towards Model-based Reinforcement Learning for Industry-near Environments","date":"2019-07-27","arxiv_id":"1907.11971","repositories_listed":1,"syntology":null},{"url":"/paper/striving-for-simplicity-in-off-policy-deep","slug":"striving-for-simplicity-in-off-policy-deep","title":"An Optimistic Perspective on Offline Reinforcement Learning","date":"2019-07-10","arxiv_id":"1907.04543","repositories_listed":1,"syntology":null},{"url":"/paper/an-intelligent-financial-portfolio-trading","slug":"an-intelligent-financial-portfolio-trading","title":"An intelligent financial portfolio trading strategy using deep Q-learning","date":"2019-07-08","arxiv_id":"1907.03665","repositories_listed":1,"syntology":null},{"url":"/paper/way-off-policy-batch-deep-reinforcement","slug":"way-off-policy-batch-deep-reinforcement","title":"Way Off-Policy Batch Deep Reinforcement Learning of Implicit Human Preferences in Dialog","date":"2019-06-30","arxiv_id":"1907.00456","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/way-off-policy-batch-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1907.00456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.00456"}},"official":{"repos":["natashamjaques/neural_chat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-empathic-deep-q-learning","slug":"towards-empathic-deep-q-learning","title":"Towards Empathic Deep Q-Learning","date":"2019-06-26","arxiv_id":"1906.10918","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-empathic-deep-q-learning#ran","syntology_url":"https://syntology.ai/paper/1906.10918","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.10918"}},"official":{"repos":["bartbussmann/EmpathicDQN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-models-of-human","slug":"reinforcement-learning-models-of-human","title":"A Story of Two Streams: Reinforcement Learning Models from Human Behavior and Neuropsychiatry","date":"2019-06-21","arxiv_id":"1906.11286","repositories_listed":1,"syntology":null},{"url":"/paper/split-q-learning-reinforcement-learning-with","slug":"split-q-learning-reinforcement-learning-with","title":"Split Q Learning: Reinforcement Learning with Two-Stream Rewards","date":"2019-06-21","arxiv_id":"1906.12350","repositories_listed":1,"syntology":null}],"record_sha256":"57ea38d95e870cca8f482cdc27299bf37ded4ee7134af19ee6e3e0dd8b6d38b2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}