{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/9","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":59,"rows_per_page":100,"rows":[801,900],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/8","next":"/task/deep-reinforcement-learning/papers/10","papers":[{"url":"/paper/example-guided-learning-of-stochastic-human","slug":"example-guided-learning-of-stochastic-human","title":"Example-guided learning of stochastic human driving policies using deep reinforcement learning","date":"2022-12-23","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/control-of-continuous-quantum-systems-with","slug":"control-of-continuous-quantum-systems-with","title":"Control of Continuous Quantum Systems with Many Degrees of Freedom based on Convergent Reinforcement Learning","date":"2022-12-21","arxiv_id":"2212.10705","repositories_listed":1,"syntology":null},{"url":"/paper/decision-making-and-control-with-metasurface","slug":"decision-making-and-control-with-metasurface","title":"Decision-making and control with diffractive optical networks","date":"2022-12-21","arxiv_id":"2212.11278","repositories_listed":1,"syntology":null},{"url":"/paper/on-reinforcement-learning-for-the-game-of","slug":"on-reinforcement-learning-for-the-game-of","title":"On Reinforcement Learning for the Game of 2048","date":"2022-12-21","arxiv_id":"2212.11087","repositories_listed":1,"syntology":null},{"url":"/paper/comparison-of-model-free-and-model-based","slug":"comparison-of-model-free-and-model-based","title":"Comparison of Model-Free and Model-Based Learning-Informed Planning for PointGoal Navigation","date":"2022-12-17","arxiv_id":"2212.08801","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-training-and-execution-multi","slug":"distributed-training-and-execution-multi","title":"Distributed-Training-and-Execution Multi-Agent Reinforcement Learning for Power Control in HetNet","date":"2022-12-15","arxiv_id":"2212.07967","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-multi-agent-deep-reinforcement","slug":"hybrid-multi-agent-deep-reinforcement","title":"Hybrid Multi-agent Deep Reinforcement Learning for Autonomous Mobility on Demand Systems","date":"2022-12-14","arxiv_id":"2212.07313","repositories_listed":1,"syntology":null},{"url":"/paper/robust-policy-optimization-in-deep","slug":"robust-policy-optimization-in-deep","title":"Robust Policy Optimization in Deep Reinforcement Learning","date":"2022-12-14","arxiv_id":"2212.07536","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-policy-optimization-in-deep#ran","syntology_url":"https://syntology.ai/paper/2212.07536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.07536"}},"official":{"repos":["vwxyzjn/cleanrl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/modem-accelerating-visual-model-based","slug":"modem-accelerating-visual-model-based","title":"MoDem: Accelerating Visual Model-Based Reinforcement Learning with Demonstrations","date":"2022-12-12","arxiv_id":"2212.05698","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/modem-accelerating-visual-model-based#ran","syntology_url":"https://syntology.ai/paper/2212.05698","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.05698"}},"official":{"repos":["facebookresearch/modem"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/optimal-planning-of-hybrid-energy-storage","slug":"optimal-planning-of-hybrid-energy-storage","title":"Optimal Planning of Hybrid Energy Storage Systems using Curtailed Renewable Energy through Deep Reinforcement Learning","date":"2022-12-12","arxiv_id":"2212.05662","repositories_listed":1,"syntology":null},{"url":"/paper/specifying-behavior-preference-with-tiered","slug":"specifying-behavior-preference-with-tiered","title":"Tiered Reward: Designing Rewards for Specification and Fast Learning of Desired Behavior","date":"2022-12-07","arxiv_id":"2212.03733","repositories_listed":1,"syntology":null},{"url":"/paper/towards-a-more-efficient-computation-of","slug":"towards-a-more-efficient-computation-of","title":"Towards a more efficient computation of individual attribute and policy contribution for post-hoc explanation of cooperative multi-agent systems using Myerson values","date":"2022-12-06","arxiv_id":"2212.03041","repositories_listed":1,"syntology":null},{"url":"/paper/what-is-the-solution-for-state-adversarial","slug":"what-is-the-solution-for-state-adversarial","title":"What is the Solution for State-Adversarial Multi-Agent Reinforcement Learning?","date":"2022-12-06","arxiv_id":"2212.02705","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/what-is-the-solution-for-state-adversarial#ran","syntology_url":"https://syntology.ai/paper/2212.02705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.02705"}},"official":{"repos":["susanbao/rmarl_code"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/l2sr-learning-to-sample-and-reconstruct-for","slug":"l2sr-learning-to-sample-and-reconstruct-for","title":"L2SR: Learning to Sample and Reconstruct for Accelerated MRI via Reinforcement Learning","date":"2022-12-05","arxiv_id":"2212.02190","repositories_listed":1,"syntology":null},{"url":"/paper/rlogist-fast-observation-strategy-on-whole","slug":"rlogist-fast-observation-strategy-on-whole","title":"RLogist: Fast Observation Strategy on Whole-slide Images with Deep Reinforcement Learning","date":"2022-12-04","arxiv_id":"2212.01737","repositories_listed":1,"syntology":null},{"url":"/paper/meshdqn-a-deep-reinforcement-learning","slug":"meshdqn-a-deep-reinforcement-learning","title":"MeshDQN: A Deep Reinforcement Learning Framework for Improving Meshes in Computational Fluid Dynamics","date":"2022-12-02","arxiv_id":"2212.01428","repositories_listed":1,"syntology":null},{"url":"/paper/stl-based-synthesis-of-feedback-controllers","slug":"stl-based-synthesis-of-feedback-controllers","title":"STL-Based Synthesis of Feedback Controllers Using Reinforcement Learning","date":"2022-12-02","arxiv_id":"2212.01022","repositories_listed":1,"syntology":null},{"url":"/paper/applying-deep-reinforcement-learning-to-the","slug":"applying-deep-reinforcement-learning-to-the","title":"Applying Deep Reinforcement Learning to the HP Model for Protein Structure Prediction","date":"2022-11-27","arxiv_id":"2211.14939","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/applying-deep-reinforcement-learning-to-the#ran","syntology_url":"https://syntology.ai/paper/2211.14939","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.14939"}},"official":{"repos":["compsoftmatterbiophysics-cityu-hk/applying-drl-to-hp-model-for-protein-structure-prediction"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/explainable-and-safe-reinforcement-learning","slug":"explainable-and-safe-reinforcement-learning","title":"Explainable and Safe Reinforcement Learning for Autonomous Air Mobility","date":"2022-11-24","arxiv_id":"2211.13474","repositories_listed":1,"syntology":null},{"url":"/paper/actively-learning-costly-reward-functions-for","slug":"actively-learning-costly-reward-functions-for","title":"Actively Learning Costly Reward Functions for Reinforcement Learning","date":"2022-11-23","arxiv_id":"2211.13260","repositories_listed":1,"syntology":null},{"url":"/paper/a-low-latency-adaptive-coding-spiking","slug":"a-low-latency-adaptive-coding-spiking","title":"A Low Latency Adaptive Coding Spiking Framework for Deep Reinforcement Learning","date":"2022-11-21","arxiv_id":"2211.11760","repositories_listed":1,"syntology":null},{"url":"/paper/tinyqmix-distributed-access-control-for-mmtc","slug":"tinyqmix-distributed-access-control-for-mmtc","title":"TinyQMIX: Distributed Access Control for mMTC via Multi-agent Reinforcement Learning","date":"2022-11-21","arxiv_id":"2211.11692","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-search-for-job-shop-scheduling","slug":"learning-to-search-for-job-shop-scheduling","title":"Deep Reinforcement Learning Guided Improvement Heuristic for Job Shop Scheduling","date":"2022-11-20","arxiv_id":"2211.10936","repositories_listed":1,"syntology":null},{"url":"/paper/pic4rl-gym-a-ros2-modular-framework-for","slug":"pic4rl-gym-a-ros2-modular-framework-for","title":"PIC4rl-gym: a ROS2 modular framework for Robots Autonomous Navigation with Deep Reinforcement Learning","date":"2022-11-19","arxiv_id":"2211.10714","repositories_listed":1,"syntology":null},{"url":"/paper/online-anomalous-subtrajectory-detection-on","slug":"online-anomalous-subtrajectory-detection-on","title":"Online Anomalous Subtrajectory Detection on Road Networks with Deep Reinforcement Learning","date":"2022-11-12","arxiv_id":"2211.08415","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-deep-reinforcement-learning-with-1","slug":"efficient-deep-reinforcement-learning-with-1","title":"Efficient Deep Reinforcement Learning with Predictive Processing Proximal Policy Optimization","date":"2022-11-11","arxiv_id":"2211.06236","repositories_listed":1,"syntology":null},{"url":"/paper/fleet-rebalancing-for-expanding-shared-e","slug":"fleet-rebalancing-for-expanding-shared-e","title":"Fleet Rebalancing for Expanding Shared e-Mobility Systems: A Multi-agent Deep Reinforcement Learning Approach","date":"2022-11-11","arxiv_id":"2211.06136","repositories_listed":1,"syntology":null},{"url":"/paper/global-and-local-analysis-of-interestingness","slug":"global-and-local-analysis-of-interestingness","title":"Global and Local Analysis of Interestingness for Competency-Aware Deep Reinforcement Learning","date":"2022-11-11","arxiv_id":"2211.06376","repositories_listed":1,"syntology":null},{"url":"/paper/deep-w-networks-solving-multi-objective","slug":"deep-w-networks-solving-multi-objective","title":"Deep W-Networks: Solving Multi-Objective Optimisation Problems With Deep Reinforcement Learning","date":"2022-11-09","arxiv_id":"2211.04813","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-sequentiality-in-reinforcement","slug":"leveraging-sequentiality-in-reinforcement","title":"Leveraging Sequentiality in Reinforcement Learning from a Single Demonstration","date":"2022-11-09","arxiv_id":"2211.04786","repositories_listed":1,"syntology":null},{"url":"/paper/diversity-based-deep-reinforcement-learning","slug":"diversity-based-deep-reinforcement-learning","title":"Diversity-based Deep Reinforcement Learning Towards Multidimensional Difficulty for Fighting Game AI","date":"2022-11-04","arxiv_id":"2211.02759","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diversity-based-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2211.02759","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.02759"}},"official":{"repos":["emily-halina/brisket"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/spatial-temporal-recurrent-reinforcement","slug":"spatial-temporal-recurrent-reinforcement","title":"Spatial-temporal recurrent reinforcement learning for autonomous ships","date":"2022-11-02","arxiv_id":"2211.01004","repositories_listed":1,"syntology":null},{"url":"/paper/operator-selection-in-adaptive-large","slug":"operator-selection-in-adaptive-large","title":"Online Control of Adaptive Large Neighborhood Search using Deep Reinforcement Learning","date":"2022-11-01","arxiv_id":"2211.00759","repositories_listed":1,"syntology":null},{"url":"/paper/erl-re-2-efficient-evolutionary-reinforcement","slug":"erl-re-2-efficient-evolutionary-reinforcement","title":"ERL-Re$^2$: Efficient Evolutionary Reinforcement Learning with Shared State Representation and Individual Policy Representation","date":"2022-10-26","arxiv_id":"2210.17375","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/erl-re-2-efficient-evolutionary-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2210.17375","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.17375"}},"official":{"repos":["yeshenpy/erl-re2"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/aacher-assorted-actor-critic-deep","slug":"aacher-assorted-actor-critic-deep","title":"AACHER: Assorted Actor-Critic Deep Reinforcement Learning with Hindsight Experience Replay","date":"2022-10-24","arxiv_id":"2210.12892","repositories_listed":1,"syntology":null},{"url":"/paper/avalon-a-benchmark-for-rl-generalization","slug":"avalon-a-benchmark-for-rl-generalization","title":"Avalon: A Benchmark for RL Generalization Using Procedurally Generated Worlds","date":"2022-10-24","arxiv_id":"2210.13417","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/avalon-a-benchmark-for-rl-generalization#ran","syntology_url":"https://syntology.ai/paper/2210.13417","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13417"}},"official":null}},{"url":"/paper/empirical-analysis-of-pga-map-elites-for","slug":"empirical-analysis-of-pga-map-elites-for","title":"Empirical analysis of PGA-MAP-Elites for Neuroevolution in Uncertain Domains","date":"2022-10-24","arxiv_id":"2210.13156","repositories_listed":1,"syntology":null},{"url":"/paper/symbolic-distillation-for-learned-tcp","slug":"symbolic-distillation-for-learned-tcp","title":"Symbolic Distillation for Learned TCP Congestion Control","date":"2022-10-24","arxiv_id":"2210.16987","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/symbolic-distillation-for-learned-tcp#ran","syntology_url":"https://syntology.ai/paper/2210.16987","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.16987"}},"official":{"repos":["vita-group/symbolicpcc"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-the-evolution-of-linear-regions","slug":"understanding-the-evolution-of-linear-regions","title":"Understanding the Evolution of Linear Regions in Deep Reinforcement Learning","date":"2022-10-24","arxiv_id":"2210.13611","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/understanding-the-evolution-of-linear-regions#ran","syntology_url":"https://syntology.ai/paper/2210.13611","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13611"}},"official":{"repos":["setarehc/deep_rl_regions"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-optimization-with-advantage","slug":"policy-optimization-with-advantage","title":"Policy Optimization with Advantage Regularization for Long-Term Fairness in Decision Systems","date":"2022-10-22","arxiv_id":"2210.12546","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/policy-optimization-with-advantage#ran","syntology_url":"https://syntology.ai/paper/2210.12546","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.12546"}},"official":{"repos":["ericyangyu/pocar"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/rate-splitting-for-intelligent-reflecting","slug":"rate-splitting-for-intelligent-reflecting","title":"Rate-Splitting for Intelligent Reflecting Surface-Aided Multiuser VR Streaming","date":"2022-10-21","arxiv_id":"2210.12191","repositories_listed":1,"syntology":null},{"url":"/paper/rmbench-benchmarking-deep-reinforcement","slug":"rmbench-benchmarking-deep-reinforcement","title":"RMBench: Benchmarking Deep Reinforcement Learning for Robotic Manipulator Control","date":"2022-10-20","arxiv_id":"2210.11262","repositories_listed":1,"syntology":null},{"url":"/paper/the-pump-scheduling-problem-a-real-world","slug":"the-pump-scheduling-problem-a-real-world","title":"The Pump Scheduling Problem: A Real-World Scenario for Reinforcement Learning","date":"2022-10-20","arxiv_id":"2210.11111","repositories_listed":1,"syntology":null},{"url":"/paper/diambra-arena-a-new-reinforcement-learning","slug":"diambra-arena-a-new-reinforcement-learning","title":"DIAMBRA Arena: a New Reinforcement Learning Platform for Research and Experimentation","date":"2022-10-19","arxiv_id":"2210.10595","repositories_listed":1,"syntology":null},{"url":"/paper/intelligent-resource-allocation-in-joint","slug":"intelligent-resource-allocation-in-joint","title":"Intelligent Resource Allocation in Joint Radar-Communication With Graph Neural Networks","date":"2022-10-17","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/cup-critic-guided-policy-reuse","slug":"cup-critic-guided-policy-reuse","title":"CUP: Critic-Guided Policy Reuse","date":"2022-10-15","arxiv_id":"2210.08153","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cup-critic-guided-policy-reuse#ran","syntology_url":"https://syntology.ai/paper/2210.08153","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.08153"}},"official":{"repos":["nagisazj/cup"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/distributional-reward-estimation-for","slug":"distributional-reward-estimation-for","title":"Distributional Reward Estimation for Effective Multi-Agent Deep Reinforcement Learning","date":"2022-10-14","arxiv_id":"2210.07636","repositories_listed":1,"syntology":null},{"url":"/paper/just-round-quantized-observation-spaces","slug":"just-round-quantized-observation-spaces","title":"Just Round: Quantized Observation Spaces Enable Memory Efficient Learning of Dynamic Locomotion","date":"2022-10-14","arxiv_id":"2210.08065","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-safe-deep-reinforcement-learning","slug":"model-based-safe-deep-reinforcement-learning","title":"Model-based Safe Deep Reinforcement Learning via a Constrained Proximal Policy Optimization Algorithm","date":"2022-10-14","arxiv_id":"2210.07573","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/model-based-safe-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2210.07573","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07573"}},"official":{"repos":["akjayant/mbppol"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/touplegdd-a-fine-designed-solution-of","slug":"touplegdd-a-fine-designed-solution-of","title":"ToupleGDD: A Fine-Designed Solution of Influence Maximization by Deep Reinforcement Learning","date":"2022-10-14","arxiv_id":"2210.07500","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/touplegdd-a-fine-designed-solution-of#ran","syntology_url":"https://syntology.ai/paper/2210.07500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07500"}},"official":{"repos":["Dtrycode/ToupleGDD"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wild-scav-benchmarking-fps-gaming-ai-on","slug":"wild-scav-benchmarking-fps-gaming-ai-on","title":"WILD-SCAV: Benchmarking FPS Gaming AI on Unity3D-based Environments","date":"2022-10-14","arxiv_id":"2210.09026","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based-rebalancing","slug":"deep-reinforcement-learning-based-rebalancing","title":"Deep Reinforcement Learning-based Rebalancing Policies for Profit Maximization of Relay Nodes in Payment Channel Networks","date":"2022-10-13","arxiv_id":"2210.07302","repositories_listed":1,"syntology":null},{"url":"/paper/harfang3d-dog-fight-sandbox-a-reinforcement","slug":"harfang3d-dog-fight-sandbox-a-reinforcement","title":"Harfang3D Dog-Fight Sandbox: A Reinforcement Learning Research Platform for the Customized Control Tasks of Fighter Aircrafts","date":"2022-10-13","arxiv_id":"2210.07282","repositories_listed":1,"syntology":null},{"url":"/paper/prosky-neat-meets-noma-mmwave-in-the-sky-of","slug":"prosky-neat-meets-noma-mmwave-in-the-sky-of","title":"ProSky: NEAT Meets NOMA-mmWave in the Sky of 6G","date":"2022-10-13","arxiv_id":"2210.11406","repositories_listed":1,"syntology":null},{"url":"/paper/towards-trustworthy-automatic-diagnosis","slug":"towards-trustworthy-automatic-diagnosis","title":"Towards Trustworthy Automatic Diagnosis Systems by Emulating Doctors' Reasoning with Deep Reinforcement Learning","date":"2022-10-13","arxiv_id":"2210.07198","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-trustworthy-automatic-diagnosis#ran","syntology_url":"https://syntology.ai/paper/2210.07198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07198"}},"official":{"repos":["mila-iqia/casande-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-adversarial-training-without","slug":"efficient-adversarial-training-without","title":"Efficient Adversarial Training without Attacking: Worst-Case-Aware Robust Reinforcement Learning","date":"2022-10-12","arxiv_id":"2210.05927","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-adversarial-training-without#ran","syntology_url":"https://syntology.ai/paper/2210.05927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05927"}},"official":{"repos":["umd-huang-lab/wocar-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-reinforcement-learning-1","slug":"benchmarking-reinforcement-learning-1","title":"Benchmarking Reinforcement Learning Techniques for Autonomous Navigation","date":"2022-10-10","arxiv_id":"2210.04839","repositories_listed":1,"syntology":{"n":6,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/benchmarking-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2210.04839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.04839"}},"official":null}},{"url":"/paper/multiagent-reinforcement-learning-based-on","slug":"multiagent-reinforcement-learning-based-on","title":"Multiagent Reinforcement Learning Based on Fusion-Multiactor-Attention-Critic for Multiple-Unmanned-Aerial-Vehicle Navigation Control","date":"2022-10-10","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/dimes-a-differentiable-meta-solver-for","slug":"dimes-a-differentiable-meta-solver-for","title":"DIMES: A Differentiable Meta Solver for Combinatorial Optimization Problems","date":"2022-10-08","arxiv_id":"2210.04123","repositories_listed":1,"syntology":{"n":25,"n_ran":10,"n_constructed":1,"n_ran_checked":1,"n_instrument":9,"n_unverified":15,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"10 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 9 where Syntology's instrument failed) · 15 unverified","sample_list":"/paper/dimes-a-differentiable-meta-solver-for#ran","syntology_url":"https://syntology.ai/paper/2210.04123","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.04123"}},"official":{"repos":["dimesteam/dimes"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":15,"ran_from_kinds":["official"]}}},{"url":"/paper/high-performance-on-atari-games-using","slug":"high-performance-on-atari-games-using","title":"High Performance on Atari Games Using Perceptual Control Architecture Without Training","date":"2022-10-08","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/flexible-attention-based-multi-policy-fusion-1","slug":"flexible-attention-based-multi-policy-fusion-1","title":"Flexible Attention-Based Multi-Policy Fusion for Efficient Deep Reinforcement Learning","date":"2022-10-07","arxiv_id":"2210.03729","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/flexible-attention-based-multi-policy-fusion-1#ran","syntology_url":"https://syntology.ai/paper/2210.03729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03729"}},"official":{"repos":["pascalson/kgrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-based-evasion","slug":"deep-reinforcement-learning-based-evasion","title":"Deep Reinforcement Learning based Evasion Generative Adversarial Network for Botnet Detection","date":"2022-10-06","arxiv_id":"2210.02840","repositories_listed":1,"syntology":null},{"url":"/paper/neuroevolution-is-a-competitive-alternative","slug":"neuroevolution-is-a-competitive-alternative","title":"Neuroevolution is a Competitive Alternative to Reinforcement Learning for Skill Discovery","date":"2022-10-06","arxiv_id":"2210.03516","repositories_listed":1,"syntology":null},{"url":"/paper/training-diverse-high-dimensional-controllers","slug":"training-diverse-high-dimensional-controllers","title":"Training Diverse High-Dimensional Controllers by Scaling Covariance Matrix Adaptation MAP-Annealing","date":"2022-10-06","arxiv_id":"2210.02622","repositories_listed":1,"syntology":null},{"url":"/paper/towards-safe-mechanical-ventilation-treatment","slug":"towards-safe-mechanical-ventilation-treatment","title":"Towards Safe Mechanical Ventilation Treatment Using Deep Offline Reinforcement Learning","date":"2022-10-05","arxiv_id":"2210.02552","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-safe-mechanical-ventilation-treatment#ran","syntology_url":"https://syntology.ai/paper/2210.02552","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.02552"}},"official":{"repos":["FlemmingKondrup/DeepVent"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/discover-deep-identification-of-symbolic-open","slug":"discover-deep-identification-of-symbolic-open","title":"DISCOVER: Deep identification of symbolically concise open-form PDEs via enhanced reinforcement-learning","date":"2022-10-04","arxiv_id":"2210.02181","repositories_listed":1,"syntology":null},{"url":"/paper/occlusion-aware-crowd-navigation-using-people","slug":"occlusion-aware-crowd-navigation-using-people","title":"Occlusion-Aware Crowd Navigation Using People as Sensors","date":"2022-10-02","arxiv_id":"2210.00552","repositories_listed":1,"syntology":null},{"url":"/paper/exploiting-transformer-in-reinforcement","slug":"exploiting-transformer-in-reinforcement","title":"Exploiting Transformer in Sparse Reward Reinforcement Learning for Interpretable Temporal Logic Motion Planning","date":"2022-09-27","arxiv_id":"2209.13220","repositories_listed":1,"syntology":null},{"url":"/paper/pretraining-the-vision-transformer-using-self","slug":"pretraining-the-vision-transformer-using-self","title":"Pretraining the Vision Transformer using self-supervised methods for vision based Deep Reinforcement Learning","date":"2022-09-22","arxiv_id":"2209.10901","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-decentralized-deep-reinforcement","slug":"hierarchical-decentralized-deep-reinforcement","title":"Hierarchical Decentralized Deep Reinforcement Learning Architecture for a Simulated Four-Legged Agent","date":"2022-09-21","arxiv_id":"2210.08003","repositories_listed":1,"syntology":null},{"url":"/paper/deep-generalized-schrodinger-bridge","slug":"deep-generalized-schrodinger-bridge","title":"Deep Generalized Schrödinger Bridge","date":"2022-09-20","arxiv_id":"2209.09893","repositories_listed":1,"syntology":{"n":9,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/deep-generalized-schrodinger-bridge#ran","syntology_url":"https://syntology.ai/paper/2209.09893","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.09893"}},"official":{"repos":["ghliu/deepgsb"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/man-multi-action-networks-learning","slug":"man-multi-action-networks-learning","title":"MAN: Multi-Action Networks Learning","date":"2022-09-19","arxiv_id":"2209.09329","repositories_listed":1,"syntology":null},{"url":"/paper/look-where-you-look-saliency-guided-q","slug":"look-where-you-look-saliency-guided-q","title":"Look where you look! Saliency-guided Q-networks for generalization in visual Reinforcement Learning","date":"2022-09-16","arxiv_id":"2209.09203","repositories_listed":1,"syntology":null},{"url":"/paper/stability-constrained-reinforcement-learning-1","slug":"stability-constrained-reinforcement-learning-1","title":"Stability Constrained Reinforcement Learning for Decentralized Real-Time Voltage Control","date":"2022-09-16","arxiv_id":"2209.07669","repositories_listed":1,"syntology":null},{"url":"/paper/toward-safe-and-accelerated-deep","slug":"toward-safe-and-accelerated-deep","title":"Toward Safe and Accelerated Deep Reinforcement Learning for Next-Generation Wireless Networks","date":"2022-09-16","arxiv_id":"2209.13532","repositories_listed":1,"syntology":null},{"url":"/paper/exploiting-reward-shifting-in-value-based","slug":"exploiting-reward-shifting-in-value-based","title":"Optimistic Curiosity Exploration and Conservative Exploitation with Linear Reward Shaping","date":"2022-09-15","arxiv_id":"2209.07288","repositories_listed":1,"syntology":null},{"url":"/paper/learning-state-correspondence-of","slug":"learning-state-correspondence-of","title":"Knowledge Transfer in Deep Reinforcement Learning via an RL-Specific GAN-Based Correspondence Function","date":"2022-09-14","arxiv_id":"2209.06604","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-solve-multiple-tsp-with-time","slug":"learning-to-solve-multiple-tsp-with-time","title":"Learning to Solve Multiple-TSP with Time Window and Rejections via Deep Reinforcement Learning","date":"2022-09-13","arxiv_id":"2209.06094","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-1","slug":"deep-reinforcement-learning-for-1","title":"Deep Reinforcement Learning for Cryptocurrency Trading: Practical Approach to Address Backtest Overfitting","date":"2022-09-12","arxiv_id":"2209.05559","repositories_listed":1,"syntology":{"n":13,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/deep-reinforcement-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/2209.05559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.05559"}},"official":null}},{"url":"/paper/partial-observability-during-drl-for-robot","slug":"partial-observability-during-drl-for-robot","title":"Experimental Study on The Effect of Multi-step Deep Reinforcement Learning in POMDPs","date":"2022-09-12","arxiv_id":"2209.04999","repositories_listed":1,"syntology":null},{"url":"/paper/reward-delay-attacks-on-deep-reinforcement","slug":"reward-delay-attacks-on-deep-reinforcement","title":"Reward Delay Attacks on Deep Reinforcement Learning","date":"2022-09-08","arxiv_id":"2209.03540","repositories_listed":1,"syntology":null},{"url":"/paper/actor-prioritized-experience-replay","slug":"actor-prioritized-experience-replay","title":"Actor Prioritized Experience Replay","date":"2022-09-01","arxiv_id":"2209.00532","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-conversational-recommendations-is","slug":"rethinking-conversational-recommendations-is","title":"Rethinking Conversational Recommendations: Is Decision Tree All You Need?","date":"2022-08-31","arxiv_id":"2208.14614","repositories_listed":1,"syntology":null},{"url":"/paper/effective-multi-user-delay-constrained","slug":"effective-multi-user-delay-constrained","title":"Effective Multi-User Delay-Constrained Scheduling with Deep Recurrent Reinforcement Learning","date":"2022-08-30","arxiv_id":"2208.14074","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-representation-learning-in-deep","slug":"unsupervised-representation-learning-in-deep","title":"Unsupervised Representation Learning in Deep Reinforcement Learning: A Review","date":"2022-08-27","arxiv_id":"2208.14226","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unsupervised-representation-learning-in-deep#ran","syntology_url":"https://syntology.ai/paper/2208.14226","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.14226"}},"official":{"repos":["nicob15/state_representation_learning_methods"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-comparison-of-reinforcement-learning-1","slug":"a-comparison-of-reinforcement-learning-1","title":"A Comparison of Reinforcement Learning Frameworks for Software Testing Tasks","date":"2022-08-25","arxiv_id":"2208.12136","repositories_listed":1,"syntology":null},{"url":"/paper/importance-prioritized-policy-distillation","slug":"importance-prioritized-policy-distillation","title":"Importance Prioritized Policy Distillation","date":"2022-08-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-risk-sensitive-approach-to-policy","slug":"a-risk-sensitive-approach-to-policy","title":"A Risk-Sensitive Approach to Policy Optimization","date":"2022-08-19","arxiv_id":"2208.09106","repositories_listed":1,"syntology":null},{"url":"/paper/a-walk-in-the-park-learning-to-walk-in-20","slug":"a-walk-in-the-park-learning-to-walk-in-20","title":"A Walk in the Park: Learning to Walk in 20 Minutes With Model-Free Reinforcement Learning","date":"2022-08-16","arxiv_id":"2208.07860","repositories_listed":1,"syntology":null},{"url":"/paper/autoshard-automated-embedding-table-sharding","slug":"autoshard-automated-embedding-table-sharding","title":"AutoShard: Automated Embedding Table Sharding for Recommender Systems","date":"2022-08-12","arxiv_id":"2208.06399","repositories_listed":1,"syntology":null},{"url":"/paper/object-detection-with-deep-reinforcement","slug":"object-detection-with-deep-reinforcement","title":"Object Detection with Deep Reinforcement Learning","date":"2022-08-09","arxiv_id":"2208.04511","repositories_listed":1,"syntology":null},{"url":"/paper/mobility-aware-cooperative-caching-in","slug":"mobility-aware-cooperative-caching-in","title":"Mobility-Aware Cooperative Caching in Vehicular Edge Computing Based on Asynchronous Federated and Deep Reinforcement Learning","date":"2022-08-02","arxiv_id":"2208.01219","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-grasp-on-the-moon-from-3d-octree","slug":"learning-to-grasp-on-the-moon-from-3d-octree","title":"Learning to Grasp on the Moon from 3D Octree Observations with Deep Reinforcement Learning","date":"2022-08-01","arxiv_id":"2208.00818","repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-correction-for-actor-critic","slug":"off-policy-correction-for-actor-critic","title":"Mitigating Off-Policy Bias in Actor-Critic Methods with One-Step Q-learning: A Novel Correction Approach","date":"2022-08-01","arxiv_id":"2208.00755","repositories_listed":1,"syntology":null},{"url":"/paper/performance-comparison-of-deep-rl-algorithms","slug":"performance-comparison-of-deep-rl-algorithms","title":"Performance Comparison of Deep RL Algorithms for Energy Systems Optimal Scheduling","date":"2022-08-01","arxiv_id":"2208.00728","repositories_listed":1,"syntology":null},{"url":"/paper/drl-m4mr-an-intelligent-multicast-routing","slug":"drl-m4mr-an-intelligent-multicast-routing","title":"DRL-M4MR: An Intelligent Multicast Routing Approach Based on DQN Deep Reinforcement Learning in SDN","date":"2022-07-31","arxiv_id":"2208.00383","repositories_listed":1,"syntology":null},{"url":"/paper/unified-automatic-control-of-vehicular","slug":"unified-automatic-control-of-vehicular","title":"Unified Automatic Control of Vehicular Systems with Reinforcement Learning","date":"2022-07-30","arxiv_id":"2208.00268","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-ucb-provably-efficient","slug":"contrastive-ucb-provably-efficient","title":"Contrastive UCB: Provably Efficient Contrastive Self-Supervised Learning in Online Reinforcement Learning","date":"2022-07-29","arxiv_id":"2207.14800","repositories_listed":1,"syntology":null},{"url":"/paper/cyclic-policy-distillation-sample-efficient","slug":"cyclic-policy-distillation-sample-efficient","title":"Cyclic Policy Distillation: Sample-Efficient Sim-to-Real Reinforcement Learning with Domain Randomization","date":"2022-07-29","arxiv_id":"2207.14561","repositories_listed":1,"syntology":null},{"url":"/paper/aadg-automatic-augmentation-for-domain","slug":"aadg-automatic-augmentation-for-domain","title":"AADG: Automatic Augmentation for Domain Generalization on Retinal Image Segmentation","date":"2022-07-27","arxiv_id":"2207.13249","repositories_listed":1,"syntology":null}],"record_sha256":"4ad2c1db6e121fd02dc80331b263d99a43b3cdb218876540e27c92b297f7b55a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}