{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/q-learning/papers/3","list_of":"/task/q-learning","task":"Q-Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":20,"rows_per_page":100,"rows":[201,300],"of":1918,"counts":{"archive_papers_tagged":1918,"with_a_code_link":463,"where_syntology_ran_a_sample":119,"not_listed_spam_title":0,"listed":1918,"listed_where_code_ran":119,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":102,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":102,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/q-learning","prev":"/task/q-learning/papers/2","next":"/task/q-learning/papers/4","papers":[{"url":"/paper/pgdqn-preference-guided-deep-q-network","slug":"pgdqn-preference-guided-deep-q-network","title":"PGDQN: Preference-Guided Deep Q-Network","date":"2023-10-03","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/pre-training-with-synthetic-data-helps","slug":"pre-training-with-synthetic-data-helps","title":"Pre-training with Synthetic Data Helps Offline Reinforcement Learning","date":"2023-10-01","arxiv_id":"2310.00771","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pre-training-with-synthetic-data-helps#ran","syntology_url":"https://syntology.ai/paper/2310.00771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00771"}},"official":{"repos":["victor-wang-902/synthetic-pretrain-rl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/counterfactual-conservative-q-learning-for-1","slug":"counterfactual-conservative-q-learning-for-1","title":"Counterfactual Conservative Q Learning for Offline Multi-agent Reinforcement Learning","date":"2023-09-22","arxiv_id":"2309.12696","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":3,"n_ran_checked":4,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":9,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/counterfactual-conservative-q-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/2309.12696","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.12696"}},"official":{"repos":["thu-rllab/CFCQL"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/reasoning-with-latent-diffusion-in-offline","slug":"reasoning-with-latent-diffusion-in-offline","title":"Reasoning with Latent Diffusion in Offline Reinforcement Learning","date":"2023-09-12","arxiv_id":"2309.06599","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/reasoning-with-latent-diffusion-in-offline#ran","syntology_url":"https://syntology.ai/paper/2309.06599","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.06599"}},"official":{"repos":["ldcq/ldcq"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-visual-tracking-and-reaching-with","slug":"learning-visual-tracking-and-reaching-with","title":"Learning Visual Tracking and Reaching with Deep Reinforcement Learning on a UR10e Robotic Arm","date":"2023-08-28","arxiv_id":"2308.14652","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-sampling-on","slug":"reinforcement-learning-for-sampling-on","title":"Reinforcement Learning for Sampling on Temporal Medical Imaging Sequences","date":"2023-08-28","arxiv_id":"2308.14946","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/reinforcement-learning-for-sampling-on#ran","syntology_url":"https://syntology.ai/paper/2308.14946","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.14946"}},"official":{"repos":["zhishenhuang/rlsamp"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/traffic-light-control-with-reinforcement","slug":"traffic-light-control-with-reinforcement","title":"Traffic Light Control with Reinforcement Learning","date":"2023-08-28","arxiv_id":"2308.14295","repositories_listed":1,"syntology":null},{"url":"/paper/towards-few-shot-coordination-revisiting-ad","slug":"towards-few-shot-coordination-revisiting-ad","title":"Towards Few-shot Coordination: Revisiting Ad-hoc Teamplay Challenge In the Game of Hanabi","date":"2023-08-20","arxiv_id":"2308.10284","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-few-shot-coordination-revisiting-ad#ran","syntology_url":"https://syntology.ai/paper/2308.10284","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.10284"}},"official":{"repos":["chandar-lab/adaptive-hanabi"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-multi-agent-reinforcement-learning-3","slug":"robust-multi-agent-reinforcement-learning-3","title":"Robust Multi-Agent Reinforcement Learning with State Uncertainty","date":"2023-07-30","arxiv_id":"2307.16212","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/robust-multi-agent-reinforcement-learning-3#ran","syntology_url":"https://syntology.ai/paper/2307.16212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.16212"}},"official":{"repos":["sihongho/robust_marl_with_state_uncertainty"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-reinforcement-learning-techniques","slug":"exploring-reinforcement-learning-techniques","title":"Exploring reinforcement learning techniques for discrete and continuous control tasks in the MuJoCo environment","date":"2023-07-20","arxiv_id":"2307.11166","repositories_listed":1,"syntology":null},{"url":"/paper/meta-value-learning-a-general-framework-for","slug":"meta-value-learning-a-general-framework-for","title":"Meta-Value Learning: a General Framework for Learning with Learning Awareness","date":"2023-07-17","arxiv_id":"2307.08863","repositories_listed":1,"syntology":null},{"url":"/paper/active-collection-of-well-being-and-health","slug":"active-collection-of-well-being-and-health","title":"Active Collection of Well-Being and Health Data in Mobile Devices","date":"2023-07-07","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/traceable-group-wise-self-optimizing-feature","slug":"traceable-group-wise-self-optimizing-feature","title":"Traceable Group-Wise Self-Optimizing Feature Transformation Learning: A Dual Optimization Perspective","date":"2023-06-29","arxiv_id":"2306.16893","repositories_listed":1,"syntology":null},{"url":"/paper/joint-path-planning-and-power-allocation-of-a","slug":"joint-path-planning-and-power-allocation-of-a","title":"Joint Path planning and Power Allocation of a Cellular-Connected UAV using Apprenticeship Learning via Deep Inverse Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.10071","repositories_listed":1,"syntology":null},{"url":"/paper/agent-performing-autonomous-stock-trading","slug":"agent-performing-autonomous-stock-trading","title":"Agent Performing Autonomous Stock Trading under Good and Bad Situations","date":"2023-06-06","arxiv_id":"2306.03985","repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-rl-algorithms-can-be-sample","slug":"off-policy-rl-algorithms-can-be-sample","title":"Off-Policy RL Algorithms Can be Sample-Efficient for Continuous Control via Sample Multiple Reuse","date":"2023-05-29","arxiv_id":"2305.18443","repositories_listed":1,"syntology":null},{"url":"/paper/madiff-offline-multi-agent-learning-with","slug":"madiff-offline-multi-agent-learning-with","title":"MADiff: Offline Multi-agent Learning with Diffusion Models","date":"2023-05-27","arxiv_id":"2305.17330","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/madiff-offline-multi-agent-learning-with#ran","syntology_url":"https://syntology.ai/paper/2305.17330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17330"}},"official":{"repos":["zbzhu99/madiff"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/2305-14550","slug":"2305-14550","title":"When should we prefer Decision Transformers for Offline Reinforcement Learning?","date":"2023-05-23","arxiv_id":"2305.14550","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/2305-14550#ran","syntology_url":"https://syntology.ai/paper/2305.14550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14550"}},"official":{"repos":["prajjwal1/rl_paradigm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mastering-percolation-like-games-with-deep","slug":"mastering-percolation-like-games-with-deep","title":"Mastering Percolation-like Games with Deep Learning","date":"2023-05-12","arxiv_id":"2305.07687","repositories_listed":1,"syntology":null},{"url":"/paper/mixed-integer-optimal-control-via","slug":"mixed-integer-optimal-control-via","title":"Mixed-Integer Optimal Control via Reinforcement Learning: A Case Study on Hybrid Electric Vehicle Energy Management","date":"2023-05-02","arxiv_id":"2305.01461","repositories_listed":1,"syntology":null},{"url":"/paper/model-free-motion-planning-of-autonomous","slug":"model-free-motion-planning-of-autonomous","title":"Model-free Motion Planning of Autonomous Agents for Complex Tasks in Partially Observable Environments","date":"2023-04-30","arxiv_id":"2305.00561","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-using-hybrid","slug":"deep-reinforcement-learning-using-hybrid","title":"Deep-Q Learning with Hybrid Quantum Neural Network on Solving Maze Problems","date":"2023-04-20","arxiv_id":"2304.10159","repositories_listed":1,"syntology":null},{"url":"/paper/idql-implicit-q-learning-as-an-actor-critic","slug":"idql-implicit-q-learning-as-an-actor-critic","title":"IDQL: Implicit Q-Learning as an Actor-Critic Method with Diffusion Policies","date":"2023-04-20","arxiv_id":"2304.10573","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/idql-implicit-q-learning-as-an-actor-critic#ran","syntology_url":"https://syntology.ai/paper/2304.10573","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.10573"}},"official":{"repos":["philippe-eecs/idql"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/collaborative-multi-bs-power-management-for","slug":"collaborative-multi-bs-power-management-for","title":"Collaborative Multi-BS Power Management for Dense Radio Access Network using Deep Reinforcement Learning","date":"2023-04-17","arxiv_id":"2304.07976","repositories_listed":1,"syntology":null},{"url":"/paper/automaton-guided-curriculum-generation-for","slug":"automaton-guided-curriculum-generation-for","title":"Automaton-Guided Curriculum Generation for Reinforcement Learning Agents","date":"2023-04-11","arxiv_id":"2304.05271","repositories_listed":1,"syntology":null},{"url":"/paper/generating-a-graph-colouring-heuristic-with","slug":"generating-a-graph-colouring-heuristic-with","title":"Generating a Graph Colouring Heuristic with Deep Q-Learning and Graph Neural Networks","date":"2023-04-08","arxiv_id":"2304.04051","repositories_listed":1,"syntology":null},{"url":"/paper/schrodinger-s-camera-first-steps-towards-a","slug":"schrodinger-s-camera-first-steps-towards-a","title":"Schrödinger's Camera: First Steps Towards a Quantum-Based Privacy Preserving Camera","date":"2023-03-13","arxiv_id":"2303.07510","repositories_listed":1,"syntology":null},{"url":"/paper/ghq-grouped-hybrid-q-learning-for","slug":"ghq-grouped-hybrid-q-learning-for","title":"GHQ: Grouped Hybrid Q Learning for Heterogeneous Cooperative Multi-agent Reinforcement Learning","date":"2023-03-02","arxiv_id":"2303.01070","repositories_listed":1,"syntology":null},{"url":"/paper/ls-iq-implicit-reward-regularization-for","slug":"ls-iq-implicit-reward-regularization-for","title":"LS-IQ: Implicit Reward Regularization for Inverse Reinforcement Learning","date":"2023-03-01","arxiv_id":"2303.00599","repositories_listed":1,"syntology":null},{"url":"/paper/potential-based-reward-shaping-for-learning","slug":"potential-based-reward-shaping-for-learning","title":"Learning to Play Text-based Adventure Games with Maximum Entropy Reinforcement Learning","date":"2023-02-21","arxiv_id":"2302.10720","repositories_listed":1,"syntology":null},{"url":"/paper/learning-from-multiple-independent-advisors","slug":"learning-from-multiple-independent-advisors","title":"Learning from Multiple Independent Advisors in Multi-agent Reinforcement Learning","date":"2023-01-26","arxiv_id":"2301.11153","repositories_listed":1,"syntology":null},{"url":"/paper/transfqmix-transformers-for-leveraging-the","slug":"transfqmix-transformers-for-leveraging-the","title":"TransfQMix: Transformers for Leveraging the Graph Structure of Multi-Agent Reinforcement Learning Problems","date":"2023-01-13","arxiv_id":"2301.05334","repositories_listed":1,"syntology":null},{"url":"/paper/training-a-deep-q-learning-agent-inside-a","slug":"training-a-deep-q-learning-agent-inside-a","title":"Learning a Generic Value-Selection Heuristic Inside a Constraint Programming Solver","date":"2023-01-05","arxiv_id":"2301.01913","repositories_listed":1,"syntology":null},{"url":"/paper/nars-vs-reinforcement-learning-ona-vs-q","slug":"nars-vs-reinforcement-learning-ona-vs-q","title":"NARS vs. Reinforcement learning: ONA vs. Q-Learning","date":"2022-12-23","arxiv_id":"2212.12517","repositories_listed":1,"syntology":null},{"url":"/paper/control-of-continuous-quantum-systems-with","slug":"control-of-continuous-quantum-systems-with","title":"Control of Continuous Quantum Systems with Many Degrees of Freedom based on Convergent Reinforcement Learning","date":"2022-12-21","arxiv_id":"2212.10705","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-training-and-execution-multi","slug":"distributed-training-and-execution-multi","title":"Distributed-Training-and-Execution Multi-Agent Reinforcement Learning for Power Control in HetNet","date":"2022-12-15","arxiv_id":"2212.07967","repositories_listed":1,"syntology":null},{"url":"/paper/policy-transfer-via-enhanced-action-space","slug":"policy-transfer-via-enhanced-action-space","title":"EASpace: Enhanced Action Space for Policy Transfer","date":"2022-12-07","arxiv_id":"2212.03540","repositories_listed":1,"syntology":null},{"url":"/paper/a-machine-with-short-term-episodic-and","slug":"a-machine-with-short-term-episodic-and","title":"A Machine with Short-Term, Episodic, and Semantic Memory Systems","date":"2022-12-05","arxiv_id":"2212.02098","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/a-machine-with-short-term-episodic-and#ran","syntology_url":"https://syntology.ai/paper/2212.02098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.02098"}},"official":{"repos":["humemai/agent-room-env-v1"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/automata-learning-meets-shielding","slug":"automata-learning-meets-shielding","title":"Automata Learning meets Shielding","date":"2022-12-04","arxiv_id":"2212.01838","repositories_listed":1,"syntology":null},{"url":"/paper/welfare-and-fairness-in-multi-objective","slug":"welfare-and-fairness-in-multi-objective","title":"Welfare and Fairness in Multi-objective Reinforcement Learning","date":"2022-11-30","arxiv_id":"2212.01382","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/welfare-and-fairness-in-multi-objective#ran","syntology_url":"https://syntology.ai/paper/2212.01382","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.01382"}},"official":{"repos":["MuhangTian/Fair-MORL-AAMAS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/ace-cooperative-multi-agent-q-learning-with","slug":"ace-cooperative-multi-agent-q-learning-with","title":"ACE: Cooperative Multi-agent Q-learning with Bidirectional Action-Dependency","date":"2022-11-29","arxiv_id":"2211.16068","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ace-cooperative-multi-agent-q-learning-with#ran","syntology_url":"https://syntology.ai/paper/2211.16068","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.16068"}},"official":{"repos":["opendilab/ace"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/explainable-and-safe-reinforcement-learning","slug":"explainable-and-safe-reinforcement-learning","title":"Explainable and Safe Reinforcement Learning for Autonomous Air Mobility","date":"2022-11-24","arxiv_id":"2211.13474","repositories_listed":1,"syntology":null},{"url":"/paper/examining-policy-entropy-of-reinforcement","slug":"examining-policy-entropy-of-reinforcement","title":"Examining Policy Entropy of Reinforcement Learning Agents for Personalization Tasks","date":"2022-11-21","arxiv_id":"2211.11869","repositories_listed":1,"syntology":null},{"url":"/paper/dynamiclight-dynamically-tuning-traffic","slug":"dynamiclight-dynamically-tuning-traffic","title":"DynamicLight: Two-Stage Dynamic Traffic Signal Timing","date":"2022-11-02","arxiv_id":"2211.01025","repositories_listed":1,"syntology":null},{"url":"/paper/solving-continuous-control-via-q-learning","slug":"solving-continuous-control-via-q-learning","title":"Solving Continuous Control via Q-learning","date":"2022-10-22","arxiv_id":"2210.12566","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/solving-continuous-control-via-q-learning#ran","syntology_url":"https://syntology.ai/paper/2210.12566","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.12566"}},"official":{"repos":["tseyde/decqn"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mutual-information-regularized-offline-1","slug":"mutual-information-regularized-offline-1","title":"Mutual Information Regularized Offline Reinforcement Learning","date":"2022-10-14","arxiv_id":"2210.07484","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mutual-information-regularized-offline-1#ran","syntology_url":"https://syntology.ai/paper/2210.07484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07484"}},"official":{"repos":["sail-sg/misa"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hybrid-rl-using-both-offline-and-online-data","slug":"hybrid-rl-using-both-offline-and-online-data","title":"Hybrid RL: Using Both Offline and Online Data Can Make RL Efficient","date":"2022-10-13","arxiv_id":"2210.06718","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hybrid-rl-using-both-offline-and-online-data#ran","syntology_url":"https://syntology.ai/paper/2210.06718","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06718"}},"official":{"repos":["yudasong/hyq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sustainable-online-reinforcement-learning-for","slug":"sustainable-online-reinforcement-learning-for","title":"Sustainable Online Reinforcement Learning for Auto-bidding","date":"2022-10-13","arxiv_id":"2210.07006","repositories_listed":1,"syntology":null},{"url":"/paper/factors-of-influence-of-the-overestimation","slug":"factors-of-influence-of-the-overestimation","title":"Factors of Influence of the Overestimation Bias of Q-Learning","date":"2022-10-11","arxiv_id":"2210.05262","repositories_listed":1,"syntology":null},{"url":"/paper/pre-training-for-robots-offline-rl-enables","slug":"pre-training-for-robots-offline-rl-enables","title":"Pre-Training for Robots: Offline RL Enables Learning New Tasks from a Handful of Trials","date":"2022-10-11","arxiv_id":"2210.05178","repositories_listed":1,"syntology":null},{"url":"/paper/towards-safe-mechanical-ventilation-treatment","slug":"towards-safe-mechanical-ventilation-treatment","title":"Towards Safe Mechanical Ventilation Treatment Using Deep Offline Reinforcement Learning","date":"2022-10-05","arxiv_id":"2210.02552","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-safe-mechanical-ventilation-treatment#ran","syntology_url":"https://syntology.ai/paper/2210.02552","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.02552"}},"official":{"repos":["FlemmingKondrup/DeepVent"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-q-learning-algorithm-for-markov","slug":"robust-q-learning-algorithm-for-markov","title":"Robust $Q$-learning Algorithm for Markov Decision Processes under Wasserstein Uncertainty","date":"2022-09-30","arxiv_id":"2210.00898","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robust-q-learning-algorithm-for-markov#ran","syntology_url":"https://syntology.ai/paper/2210.00898","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.00898"}},"official":{"repos":["juliansester/wasserstein-q-learning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-discrete-soft-actor-critic","slug":"revisiting-discrete-soft-actor-critic","title":"Revisiting Discrete Soft Actor-Critic","date":"2022-09-21","arxiv_id":"2209.10081","repositories_listed":1,"syntology":null},{"url":"/paper/man-multi-action-networks-learning","slug":"man-multi-action-networks-learning","title":"MAN: Multi-Action Networks Learning","date":"2022-09-19","arxiv_id":"2209.09329","repositories_listed":1,"syntology":null},{"url":"/paper/m-2-dqn-a-robust-method-for-accelerating-deep","slug":"m-2-dqn-a-robust-method-for-accelerating-deep","title":"M$^2$DQN: A Robust Method for Accelerating Deep Q-learning Network","date":"2022-09-16","arxiv_id":"2209.07809","repositories_listed":1,"syntology":null},{"url":"/paper/q-learning-decision-transformer-leveraging","slug":"q-learning-decision-transformer-leveraging","title":"Q-learning Decision Transformer: Leveraging Dynamic Programming for Conditional Sequence Modelling in Offline RL","date":"2022-09-08","arxiv_id":"2209.03993","repositories_listed":1,"syntology":null},{"url":"/paper/reward-delay-attacks-on-deep-reinforcement","slug":"reward-delay-attacks-on-deep-reinforcement","title":"Reward Delay Attacks on Deep Reinforcement Learning","date":"2022-09-08","arxiv_id":"2209.03540","repositories_listed":1,"syntology":null},{"url":"/paper/goal-conditioned-q-learning-as-knowledge","slug":"goal-conditioned-q-learning-as-knowledge","title":"Goal-Conditioned Q-Learning as Knowledge Distillation","date":"2022-08-28","arxiv_id":"2208.13298","repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-correction-for-actor-critic","slug":"off-policy-correction-for-actor-critic","title":"Mitigating Off-Policy Bias in Actor-Critic Methods with One-Step Q-learning: A Novel Correction Approach","date":"2022-08-01","arxiv_id":"2208.00755","repositories_listed":1,"syntology":null},{"url":"/paper/reactive-exploration-to-cope-with-non","slug":"reactive-exploration-to-cope-with-non","title":"Reactive Exploration to Cope with Non-Stationarity in Lifelong Reinforcement Learning","date":"2022-07-12","arxiv_id":"2207.05742","repositories_listed":1,"syntology":null},{"url":"/paper/reinforced-lin-kernighan-helsgaun-algorithms","slug":"reinforced-lin-kernighan-helsgaun-algorithms","title":"Reinforced Lin-Kernighan-Helsgaun Algorithms for the Traveling Salesman Problems","date":"2022-07-08","arxiv_id":"2207.03876","repositories_listed":1,"syntology":null},{"url":"/paper/maser-multi-agent-reinforcement-learning-with","slug":"maser-multi-agent-reinforcement-learning-with","title":"MASER: Multi-Agent Reinforcement Learning with Subgoals Generated from Experience Replay Buffer","date":"2022-06-20","arxiv_id":"2206.10607","repositories_listed":1,"syntology":null},{"url":"/paper/sampling-efficient-deep-reinforcement","slug":"sampling-efficient-deep-reinforcement","title":"Sampling Efficient Deep Reinforcement Learning through Preference-Guided Stochastic Exploration","date":"2022-06-20","arxiv_id":"2206.09627","repositories_listed":1,"syntology":null},{"url":"/paper/search-based-testing-approach-for-deep","slug":"search-based-testing-approach-for-deep","title":"A Search-Based Testing Approach for Deep Reinforcement Learning Agents","date":"2022-06-15","arxiv_id":"2206.07813","repositories_listed":1,"syntology":null},{"url":"/paper/cooperation-between-independent-market-makers","slug":"cooperation-between-independent-market-makers","title":"Cooperation between Independent Market Makers","date":"2022-06-11","arxiv_id":"2206.05410","repositories_listed":1,"syntology":null},{"url":"/paper/deeptpi-test-point-insertion-with-deep","slug":"deeptpi-test-point-insertion-with-deep","title":"DeepTPI: Test Point Insertion with Deep Reinforcement Learning","date":"2022-06-07","arxiv_id":"2206.06975","repositories_listed":1,"syntology":null},{"url":"/paper/look-back-when-surprised-stabilizing-reverse","slug":"look-back-when-surprised-stabilizing-reverse","title":"Introspective Experience Replay: Look Back When Surprised","date":"2022-06-07","arxiv_id":"2206.03171","repositories_listed":1,"syntology":null},{"url":"/paper/graph-backup-data-efficient-backup-exploiting","slug":"graph-backup-data-efficient-backup-exploiting","title":"Graph Backup: Data Efficient Backup Exploiting Markovian Transitions","date":"2022-05-31","arxiv_id":"2205.15824","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-multi-class","slug":"deep-reinforcement-learning-for-multi-class","title":"Deep Reinforcement Learning for Multi-class Imbalanced Training","date":"2022-05-24","arxiv_id":"2205.12070","repositories_listed":1,"syntology":null},{"url":"/paper/simultaneous-double-q-learning-with","slug":"simultaneous-double-q-learning-with","title":"Simultaneous Double Q-learning with Conservative Advantage Learning for Actor-Critic Methods","date":"2022-05-08","arxiv_id":"2205.03819","repositories_listed":1,"syntology":null},{"url":"/paper/cclf-a-contrastive-curiosity-driven-learning","slug":"cclf-a-contrastive-curiosity-driven-learning","title":"CCLF: A Contrastive-Curiosity-Driven Learning Framework for Sample-Efficient Reinforcement Learning","date":"2022-05-02","arxiv_id":"2205.00943","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/cclf-a-contrastive-curiosity-driven-learning#ran","syntology_url":"https://syntology.ai/paper/2205.00943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.00943"}},"official":{"repos":["csun001/cclf"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/gail-pt-a-generic-intelligent-penetration","slug":"gail-pt-a-generic-intelligent-penetration","title":"GAIL-PT: A Generic Intelligent Penetration Testing Framework with Generative Adversarial Imitation Learning","date":"2022-04-05","arxiv_id":"2204.01975","repositories_listed":1,"syntology":null},{"url":"/paper/topological-experience-replay-1","slug":"topological-experience-replay-1","title":"Topological Experience Replay","date":"2022-03-29","arxiv_id":"2203.15845","repositories_listed":1,"syntology":null},{"url":"/paper/intelligent-masking-deep-q-learning-for","slug":"intelligent-masking-deep-q-learning-for","title":"Intelligent Masking: Deep Q-Learning for Context Encoding in Medical Image Analysis","date":"2022-03-25","arxiv_id":"2203.13865","repositories_listed":1,"syntology":null},{"url":"/paper/action-candidate-driven-clipped-double-q","slug":"action-candidate-driven-clipped-double-q","title":"Action Candidate Driven Clipped Double Q-learning for Discrete and Continuous Action Tasks","date":"2022-03-22","arxiv_id":"2203.11526","repositories_listed":1,"syntology":null},{"url":"/paper/orchestrated-value-mapping-for-reinforcement-1","slug":"orchestrated-value-mapping-for-reinforcement-1","title":"Orchestrated Value Mapping for Reinforcement Learning","date":"2022-03-14","arxiv_id":"2203.07171","repositories_listed":1,"syntology":null},{"url":"/paper/goal-recognition-as-reinforcement-learning","slug":"goal-recognition-as-reinforcement-learning","title":"Goal Recognition as Reinforcement Learning","date":"2022-02-13","arxiv_id":"2202.06356","repositories_listed":1,"syntology":null},{"url":"/paper/microservice-deployment-in-edge-computing","slug":"microservice-deployment-in-edge-computing","title":"Microservice Deployment in Edge Computing Based on Deep Q Learning","date":"2022-02-11","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/deep-q-learning-a-robust-control-approach","slug":"deep-q-learning-a-robust-control-approach","title":"Deep Q-learning: a robust control approach","date":"2022-01-21","arxiv_id":"2201.08610","repositories_listed":1,"syntology":null},{"url":"/paper/two-sample-testing-in-reinforcement-learning","slug":"two-sample-testing-in-reinforcement-learning","title":"Addressing Maximization Bias in Reinforcement Learning with Two-Sample Testing","date":"2022-01-20","arxiv_id":"2201.08078","repositories_listed":1,"syntology":null},{"url":"/paper/polyak-ruppert-averaged-q-leaning-is","slug":"polyak-ruppert-averaged-q-leaning-is","title":"A Statistical Analysis of Polyak-Ruppert Averaged Q-learning","date":"2021-12-29","arxiv_id":"2112.14582","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/polyak-ruppert-averaged-q-leaning-is#ran","syntology_url":"https://syntology.ai/paper/2112.14582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.14582"}},"official":{"repos":["lx10077/AveQLearning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/task-and-model-agnostic-adversarial-attack-on","slug":"task-and-model-agnostic-adversarial-attack-on","title":"Task and Model Agnostic Adversarial Attack on Graph Neural Networks","date":"2021-12-25","arxiv_id":"2112.13267","repositories_listed":1,"syntology":null},{"url":"/paper/safety-and-liveness-guarantees-through-reach","slug":"safety-and-liveness-guarantees-through-reach","title":"Safety and Liveness Guarantees through Reach-Avoid Reinforcement Learning","date":"2021-12-23","arxiv_id":"2112.12288","repositories_listed":1,"syntology":null},{"url":"/paper/regularized-softmax-deep-multi-agent-q","slug":"regularized-softmax-deep-multi-agent-q","title":"Regularized Softmax Deep Multi-Agent Q-Learning","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/continuous-control-with-ensemble-deep-1","slug":"continuous-control-with-ensemble-deep-1","title":"Continuous Control With Ensemble Deep Deterministic Policy Gradients","date":"2021-11-30","arxiv_id":"2111.15382","repositories_listed":1,"syntology":null},{"url":"/paper/solving-reward-collecting-problems-with-uavs","slug":"solving-reward-collecting-problems-with-uavs","title":"Solving reward-collecting problems with UAVs: a comparison of online optimization and Q-learning","date":"2021-11-30","arxiv_id":"2112.00141","repositories_listed":1,"syntology":null},{"url":"/paper/deep-q-learning-based-reinforcement-learning","slug":"deep-q-learning-based-reinforcement-learning","title":"Deep Q-Learning based Reinforcement Learning Approach for Network Intrusion Detection","date":"2021-11-27","arxiv_id":"2111.13978","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-the-impact-of-data-distribution","slug":"understanding-the-impact-of-data-distribution","title":"The Impact of Data Distribution on Q-learning with Function Approximation","date":"2021-11-23","arxiv_id":"2111.11758","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-advisor-q-learning","slug":"multi-agent-advisor-q-learning","title":"Multi-Agent Advisor Q-Learning","date":"2021-10-26","arxiv_id":"2111.00345","repositories_listed":1,"syntology":null},{"url":"/paper/playing-2048-with-reinforcement-learning","slug":"playing-2048-with-reinforcement-learning","title":"Playing 2048 With Reinforcement Learning","date":"2021-10-20","arxiv_id":"2110.10374","repositories_listed":1,"syntology":null},{"url":"/paper/balancing-value-underestimation-and","slug":"balancing-value-underestimation-and","title":"Balancing Value Underestimation and Overestimation with Realistic Actor-Critic","date":"2021-10-19","arxiv_id":"2110.09712","repositories_listed":1,"syntology":null},{"url":"/paper/training-transition-policies-via-distribution-1","slug":"training-transition-policies-via-distribution-1","title":"Training Transition Policies via Distribution Matching for Complex Tasks","date":"2021-10-08","arxiv_id":"2110.04357","repositories_listed":1,"syntology":null},{"url":"/paper/learning-the-markov-decision-process-in-the","slug":"learning-the-markov-decision-process-in-the","title":"Learning the Markov Decision Process in the Sparse Gaussian Elimination","date":"2021-09-30","arxiv_id":"2109.14929","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-q-learning-for-intelligent","slug":"deep-reinforcement-q-learning-for-intelligent","title":"Deep Reinforcement Q-Learning for Intelligent Traffic Signal Control with Partial Detection","date":"2021-09-29","arxiv_id":"2109.14337","repositories_listed":1,"syntology":null},{"url":"/paper/offline-reinforcement-learning-with-in-sample","slug":"offline-reinforcement-learning-with-in-sample","title":"Offline Reinforcement Learning with In-sample Q-Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/on-the-estimation-bias-in-double-q-learning-1","slug":"on-the-estimation-bias-in-double-q-learning-1","title":"On the Estimation Bias in Double Q-Learning","date":"2021-09-29","arxiv_id":"2109.14419","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/on-the-estimation-bias-in-double-q-learning-1#ran","syntology_url":"https://syntology.ai/paper/2109.14419","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.14419"}},"official":{"repos":["stilwell-git/doubly-bounded-q-learning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/parameter-free-deterministic-reduction-of-the","slug":"parameter-free-deterministic-reduction-of-the","title":"Parameter-free Reduction of the Estimation Bias in Deep Reinforcement Learning for Deterministic Policy Gradients","date":"2021-09-24","arxiv_id":"2109.11788","repositories_listed":1,"syntology":null},{"url":"/paper/estimation-error-correction-in-deep","slug":"estimation-error-correction-in-deep","title":"Estimation Error Correction in Deep Reinforcement Learning for Deterministic Actor-Critic Methods","date":"2021-09-22","arxiv_id":"2109.10736","repositories_listed":1,"syntology":null},{"url":"/paper/bootstrapped-meta-learning","slug":"bootstrapped-meta-learning","title":"Bootstrapped Meta-Learning","date":"2021-09-09","arxiv_id":"2109.04504","repositories_listed":1,"syntology":null},{"url":"/paper/deep-active-inference-for-pixel-based","slug":"deep-active-inference-for-pixel-based","title":"Deep Active Inference for Pixel-Based Discrete Control: Evaluation on the Car Racing Problem","date":"2021-09-09","arxiv_id":"2109.04155","repositories_listed":1,"syntology":null}],"record_sha256":"318b5a2c336882491b9f81e51604ecf095257a3a7edc07f2656da2bedd541b92","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}