{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/11","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":11,"pages_in_order":59,"rows_per_page":100,"rows":[1001,1100],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/10","next":"/task/deep-reinforcement-learning/papers/12","papers":[{"url":"/paper/dns-determinantal-point-process-based-neural","slug":"dns-determinantal-point-process-based-neural","title":"DNS: Determinantal Point Process Based Neural Network Sampler for Ensemble Reinforcement Learning","date":"2022-01-31","arxiv_id":"2201.13357","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dns-determinantal-point-process-based-neural#ran","syntology_url":"https://syntology.ai/paper/2201.13357","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.13357"}},"official":{"repos":["IntelLabs/DNS"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/graph-convolution-based-deep-reinforcement","slug":"graph-convolution-based-deep-reinforcement","title":"Graph Convolution-Based Deep Reinforcement Learning for Multi-Agent Decision-Making in Mixed Traffic Environments","date":"2022-01-30","arxiv_id":"2201.12776","repositories_listed":1,"syntology":null},{"url":"/paper/mask-based-latent-reconstruction-for","slug":"mask-based-latent-reconstruction-for","title":"Mask-based Latent Reconstruction for Reinforcement Learning","date":"2022-01-28","arxiv_id":"2201.12096","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":4,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 4 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mask-based-latent-reconstruction-for#ran","syntology_url":"https://syntology.ai/paper/2201.12096","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12096"}},"official":{"repos":["microsoft/Mask-based-Latent-Reconstruction"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/the-first-ai4tsp-competition-learning-to","slug":"the-first-ai4tsp-competition-learning-to","title":"The First AI4TSP Competition: Learning to Solve Stochastic Routing Problems","date":"2022-01-25","arxiv_id":"2201.10453","repositories_listed":1,"syntology":null},{"url":"/paper/environment-generation-for-zero-shot-1","slug":"environment-generation-for-zero-shot-1","title":"Environment Generation for Zero-Shot Compositional Reinforcement Learning","date":"2022-01-21","arxiv_id":"2201.08896","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/environment-generation-for-zero-shot-1#ran","syntology_url":"https://syntology.ai/paper/2201.08896","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.08896"}},"official":{"repos":["google-research/google-research"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-textbook","slug":"reinforcement-learning-textbook","title":"Reinforcement Learning Textbook","date":"2022-01-19","arxiv_id":"2201.09746","repositories_listed":1,"syntology":null},{"url":"/paper/modeling-bounded-rationality-in-multi-agent-1","slug":"modeling-bounded-rationality-in-multi-agent-1","title":"Solving Dynamic Principal-Agent Problems with a Rationally Inattentive Principal","date":"2022-01-18","arxiv_id":"2202.01691","repositories_listed":1,"syntology":null},{"url":"/paper/smart-magnetic-microrobots-learn-to-swim-with","slug":"smart-magnetic-microrobots-learn-to-swim-with","title":"Smart Magnetic Microrobots Learn to Swim with Deep Reinforcement Learning","date":"2022-01-14","arxiv_id":"2201.05599","repositories_listed":1,"syntology":null},{"url":"/paper/solving-dynamic-graph-problems-with-multi","slug":"solving-dynamic-graph-problems-with-multi","title":"Solving Dynamic Graph Problems with Multi-Attention Deep Reinforcement Learning","date":"2022-01-13","arxiv_id":"2201.04895","repositories_listed":1,"syntology":null},{"url":"/paper/weakly-supervised-scene-text-detection-using","slug":"weakly-supervised-scene-text-detection-using","title":"Weakly Supervised Scene Text Detection using Deep Reinforcement Learning","date":"2022-01-13","arxiv_id":"2201.04866","repositories_listed":1,"syntology":null},{"url":"/paper/verified-probabilistic-policies-for-deep","slug":"verified-probabilistic-policies-for-deep","title":"Verified Probabilistic Policies for Deep Reinforcement Learning","date":"2022-01-10","arxiv_id":"2201.03698","repositories_listed":1,"syntology":null},{"url":"/paper/mirror-learning-a-unifying-framework-of","slug":"mirror-learning-a-unifying-framework-of","title":"Mirror Learning: A Unifying Framework of Policy Optimisation","date":"2022-01-07","arxiv_id":"2201.02373","repositories_listed":1,"syntology":null},{"url":"/paper/balsa-learning-a-query-optimizer-without","slug":"balsa-learning-a-query-optimizer-without","title":"Balsa: Learning a Query Optimizer Without Expert Demonstrations","date":"2022-01-05","arxiv_id":"2201.01441","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-deep-reinforcement-learning-5","slug":"sample-efficient-deep-reinforcement-learning-5","title":"Sample Efficient Deep Reinforcement Learning via Uncertainty Estimation","date":"2022-01-05","arxiv_id":"2201.01666","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sample-efficient-deep-reinforcement-learning-5#ran","syntology_url":"https://syntology.ai/paper/2201.01666","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.01666"}},"official":{"repos":["montrealrobotics/iv_rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hybrid-intelligence-for-dynamic-job-shop","slug":"hybrid-intelligence-for-dynamic-job-shop","title":"Hybrid intelligence for dynamic job-shop scheduling with deep reinforcement learning and attention mechanism","date":"2022-01-03","arxiv_id":"2201.00548","repositories_listed":1,"syntology":null},{"url":"/paper/self-reward-design-with-fine-grained-1","slug":"self-reward-design-with-fine-grained-1","title":"Self Reward Design with Fine-grained Interpretability","date":"2021-12-30","arxiv_id":"2112.15034","repositories_listed":1,"syntology":null},{"url":"/paper/sequential-episodic-control","slug":"sequential-episodic-control","title":"Sequential memory improves sample and memory efficiency in Episodic Control","date":"2021-12-29","arxiv_id":"2112.14734","repositories_listed":1,"syntology":null},{"url":"/paper/intelligent-traffic-light-via-policy-based","slug":"intelligent-traffic-light-via-policy-based","title":"Intelligent Traffic Light via Policy-based Deep Reinforcement Learning","date":"2021-12-27","arxiv_id":"2112.13817","repositories_listed":1,"syntology":null},{"url":"/paper/an-efficient-combinatorial-optimization-model","slug":"an-efficient-combinatorial-optimization-model","title":"An Efficient Combinatorial Optimization Model Using Learning-to-Rank Distillation","date":"2021-12-24","arxiv_id":"2201.00695","repositories_listed":1,"syntology":null},{"url":"/paper/safety-and-liveness-guarantees-through-reach","slug":"safety-and-liveness-guarantees-through-reach","title":"Safety and Liveness Guarantees through Reach-Avoid Reinforcement Learning","date":"2021-12-23","arxiv_id":"2112.12288","repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-reinforcement-learning-approach-for-11","slug":"a-deep-reinforcement-learning-approach-for-11","title":"A Deep Reinforcement Learning Approach for Solving the Traveling Salesman Problem with Drone","date":"2021-12-22","arxiv_id":"2112.12545","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-deep-reinforcement-learning-for","slug":"adversarial-deep-reinforcement-learning-for","title":"Adversarial Deep Reinforcement Learning for Improving the Robustness of Multi-agent Autonomous Driving Policies","date":"2021-12-22","arxiv_id":"2112.11937","repositories_listed":1,"syntology":null},{"url":"/paper/alpha-mini-minichess-agent-with-deep","slug":"alpha-mini-minichess-agent-with-deep","title":"Alpha-Mini: Minichess Agent with Deep Reinforcement Learning","date":"2021-12-22","arxiv_id":"2112.13666","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-the-robustness-of-deep","slug":"evaluating-the-robustness-of-deep","title":"Evaluating the Robustness of Deep Reinforcement Learning for Autonomous Policies in a Multi-agent Urban Driving Environment","date":"2021-12-22","arxiv_id":"2112.11947","repositories_listed":1,"syntology":null},{"url":"/paper/newsvendor-model-with-deep-reinforcement","slug":"newsvendor-model-with-deep-reinforcement","title":"Newsvendor Model with Deep Reinforcement Learning","date":"2021-12-22","arxiv_id":"2112.12544","repositories_listed":1,"syntology":null},{"url":"/paper/value-activation-for-bias-alleviation","slug":"value-activation-for-bias-alleviation","title":"Value Activation for Bias Alleviation: Generalized-activated Deep Double Deterministic Policy Gradients","date":"2021-12-21","arxiv_id":"2112.11216","repositories_listed":1,"syntology":null},{"url":"/paper/space-non-cooperative-object-active-tracking","slug":"space-non-cooperative-object-active-tracking","title":"Space Non-cooperative Object Active Tracking with Deep Reinforcement Learning","date":"2021-12-18","arxiv_id":"2112.09854","repositories_listed":1,"syntology":null},{"url":"/paper/colo-ran-developing-machine-learning-based","slug":"colo-ran-developing-machine-learning-based","title":"ColO-RAN: Developing Machine Learning-based xApps for Open RAN Closed-loop Control on Programmable Experimental Platforms","date":"2021-12-17","arxiv_id":"2112.09559","repositories_listed":1,"syntology":null},{"url":"/paper/towards-disturbance-free-visual-mobile","slug":"towards-disturbance-free-visual-mobile","title":"Towards Disturbance-Free Visual Mobile Manipulation","date":"2021-12-17","arxiv_id":"2112.12612","repositories_listed":1,"syntology":null},{"url":"/paper/stochastic-actor-executor-critic-for-image-to","slug":"stochastic-actor-executor-critic-for-image-to","title":"Stochastic Actor-Executor-Critic for Image-to-Image Translation","date":"2021-12-14","arxiv_id":"2112.07403","repositories_listed":1,"syntology":null},{"url":"/paper/finrl-meta-a-universe-of-near-real-market","slug":"finrl-meta-a-universe-of-near-real-market","title":"FinRL-Meta: A Universe of Near-Real Market Environments for Data-Driven Deep Reinforcement Learning in Quantitative Finance","date":"2021-12-13","arxiv_id":"2112.06753","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/finrl-meta-a-universe-of-near-real-market#ran","syntology_url":"https://syntology.ai/paper/2112.06753","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.06753"}},"official":{"repos":["ai4finance-foundation/finrl-meta"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/elegantrl-podracer-scalable-and-elastic","slug":"elegantrl-podracer-scalable-and-elastic","title":"ElegantRL-Podracer: Scalable and Elastic Library for Cloud-Native Deep Reinforcement Learning","date":"2021-12-11","arxiv_id":"2112.05923","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/elegantrl-podracer-scalable-and-elastic#ran","syntology_url":"https://syntology.ai/paper/2112.05923","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.05923"}},"official":{"repos":["ai4finance-foundation/elegantrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-q-network-with-proximal-iteration-1","slug":"deep-q-network-with-proximal-iteration-1","title":"Faster Deep Reinforcement Learning with Slower Online Network","date":"2021-12-10","arxiv_id":"2112.05848","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-q-network-with-proximal-iteration-1#ran","syntology_url":"https://syntology.ai/paper/2112.05848","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.05848"}},"official":{"repos":["amazon-research/fast-rl-with-slow-updates"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/attention-based-model-and-deep-reinforcement","slug":"attention-based-model-and-deep-reinforcement","title":"Attention-Based Model and Deep Reinforcement Learning for Distribution of Event Processing Tasks","date":"2021-12-07","arxiv_id":"2112.03835","repositories_listed":1,"syntology":null},{"url":"/paper/federated-deep-reinforcement-learning-for-the","slug":"federated-deep-reinforcement-learning-for-the","title":"Federated Deep Reinforcement Learning for the Distributed Control of NextG Wireless Networks","date":"2021-12-07","arxiv_id":"2112.03465","repositories_listed":1,"syntology":null},{"url":"/paper/flexible-option-learning-1","slug":"flexible-option-learning-1","title":"Flexible Option Learning","date":"2021-12-06","arxiv_id":"2112.03097","repositories_listed":1,"syntology":null},{"url":"/paper/functional-regularization-for-reinforcement-1","slug":"functional-regularization-for-reinforcement-1","title":"Functional Regularization for Reinforcement Learning via Learned Fourier Features","date":"2021-12-06","arxiv_id":"2112.03257","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/functional-regularization-for-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2112.03257","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.03257"}},"official":{"repos":["alexlioralexli/learned-fourier-features"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/virtual-replay-cache","slug":"virtual-replay-cache","title":"Virtual Replay Cache","date":"2021-12-06","arxiv_id":"2112.03421","repositories_listed":1,"syntology":null},{"url":"/paper/sparrl-graph-sparsification-via-deep-1","slug":"sparrl-graph-sparsification-via-deep-1","title":"A Generic Graph Sparsification Framework using Deep Reinforcement Learning","date":"2021-12-02","arxiv_id":"2112.01565","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-data-augmentation-for-1","slug":"automatic-data-augmentation-for-1","title":"Automatic Data Augmentation for Generalization in Reinforcement Learning","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/edge-explaining-deep-reinforcement-learning","slug":"edge-explaining-deep-reinforcement-learning","title":"EDGE: Explaining Deep Reinforcement Learning Policies","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/noveld-a-simple-yet-effective-exploration","slug":"noveld-a-simple-yet-effective-exploration","title":"NovelD: A Simple yet Effective Exploration Criterion","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/symbolic-regression-via-deep-reinforcement","slug":"symbolic-regression-via-deep-reinforcement","title":"Symbolic Regression via Deep Reinforcement Learning Enhanced Genetic Programming Seeding","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/continuous-control-with-ensemble-deep-1","slug":"continuous-control-with-ensemble-deep-1","title":"Continuous Control With Ensemble Deep Deterministic Policy Gradients","date":"2021-11-30","arxiv_id":"2111.15382","repositories_listed":1,"syntology":null},{"url":"/paper/adaptively-calibrated-critic-estimates-for","slug":"adaptively-calibrated-critic-estimates-for","title":"Adaptively Calibrated Critic Estimates for Deep Reinforcement Learning","date":"2021-11-24","arxiv_id":"2111.12673","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptively-calibrated-critic-estimates-for#ran","syntology_url":"https://syntology.ai/paper/2111.12673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.12673"}},"official":{"repos":["nicolinho/acc"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/component-transfer-learning-for-deep-rl-based","slug":"component-transfer-learning-for-deep-rl-based","title":"Component Transfer Learning for Deep RL Based on Abstract Representations","date":"2021-11-22","arxiv_id":"2111.11525","repositories_listed":1,"syntology":null},{"url":"/paper/real-world-dexterous-object-manipulation","slug":"real-world-dexterous-object-manipulation","title":"Real-World Dexterous Object Manipulation based Deep Reinforcement Learning","date":"2021-11-22","arxiv_id":"2112.04893","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforced-attention-regression-for","slug":"deep-reinforced-attention-regression-for","title":"Deep Reinforced Attention Regression for Partial Sketch Based Image Retrieval","date":"2021-11-21","arxiv_id":"2111.10917","repositories_listed":1,"syntology":null},{"url":"/paper/recursive-self-improvement-for-camera-image","slug":"recursive-self-improvement-for-camera-image","title":"Reinforcement Learning of Self Enhancing Camera Image and Signal Processing","date":"2021-11-15","arxiv_id":"2111.07499","repositories_listed":1,"syntology":null},{"url":"/paper/user-allocation-in-mobile-edge-computing-a","slug":"user-allocation-in-mobile-edge-computing-a","title":"User Allocation in Mobile Edge Computing: A Deep Reinforcement Learning Approach","date":"2021-11-11","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/data-efficient-deep-reinforcement-learning-1","slug":"data-efficient-deep-reinforcement-learning-1","title":"Data-Efficient Deep Reinforcement Learning for Attitude Control of Fixed-Wing UAVs: Field Experiments","date":"2021-11-07","arxiv_id":"2111.04153","repositories_listed":1,"syntology":null},{"url":"/paper/robust-deep-reinforcement-learning-for-1","slug":"robust-deep-reinforcement-learning-for-1","title":"Robust Deep Reinforcement Learning for Quadcopter Control","date":"2021-11-06","arxiv_id":"2111.03915","repositories_listed":1,"syntology":null},{"url":"/paper/human-level-control-without-server-grade-1","slug":"human-level-control-without-server-grade-1","title":"Human-Level Control without Server-Grade Hardware","date":"2021-11-01","arxiv_id":"2111.01264","repositories_listed":1,"syntology":null},{"url":"/paper/learning-large-neighborhood-search-policy-for","slug":"learning-large-neighborhood-search-policy-for","title":"Learning Large Neighborhood Search Policy for Integer Programming","date":"2021-11-01","arxiv_id":"2111.03466","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-large-neighborhood-search-policy-for#ran","syntology_url":"https://syntology.ai/paper/2111.03466","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.03466"}},"official":{"repos":["wxy1427/learn-lns-policy"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/urlb-unsupervised-reinforcement-learning","slug":"urlb-unsupervised-reinforcement-learning","title":"URLB: Unsupervised Reinforcement Learning Benchmark","date":"2021-10-28","arxiv_id":"2110.15191","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/urlb-unsupervised-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2110.15191","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.15191"}},"official":{"repos":["rll-research/url_benchmark"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/learning-domain-invariant-representations-in","slug":"learning-domain-invariant-representations-in","title":"Learning Domain Invariant Representations in Goal-conditioned Block MDPs","date":"2021-10-27","arxiv_id":"2110.14248","repositories_listed":1,"syntology":null},{"url":"/paper/towards-robust-bisimulation-metric-learning","slug":"towards-robust-bisimulation-metric-learning","title":"Towards Robust Bisimulation Metric Learning","date":"2021-10-27","arxiv_id":"2110.14096","repositories_listed":1,"syntology":null},{"url":"/paper/learning-collaborative-policies-to-solve-np","slug":"learning-collaborative-policies-to-solve-np","title":"Learning Collaborative Policies to Solve NP-hard Routing Problems","date":"2021-10-26","arxiv_id":"2110.13987","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":4,"n_ran_checked":4,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":11,"phrase":"6 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/learning-collaborative-policies-to-solve-np#ran","syntology_url":"https://syntology.ai/paper/2110.13987","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.13987"}},"official":{"repos":["alstn12088/lcp"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/the-difficulty-of-passive-learning-in-deep","slug":"the-difficulty-of-passive-learning-in-deep","title":"The Difficulty of Passive Learning in Deep Reinforcement Learning","date":"2021-10-26","arxiv_id":"2110.14020","repositories_listed":1,"syntology":null},{"url":"/paper/recurrent-off-policy-baselines-for-memory","slug":"recurrent-off-policy-baselines-for-memory","title":"Recurrent Off-policy Baselines for Memory-based Continuous Control","date":"2021-10-25","arxiv_id":"2110.12628","repositories_listed":1,"syntology":null},{"url":"/paper/safely-bridging-offline-and-online","slug":"safely-bridging-offline-and-online","title":"Uniformly Conservative Exploration in Reinforcement Learning","date":"2021-10-25","arxiv_id":"2110.13060","repositories_listed":1,"syntology":null},{"url":"/paper/relace-reinforcement-learning-agent-for","slug":"relace-reinforcement-learning-agent-for","title":"ReLAX: Reinforcement Learning Agent eXplainer for Arbitrary Predictive Models","date":"2021-10-22","arxiv_id":"2110.11960","repositories_listed":1,"syntology":null},{"url":"/paper/balancing-value-underestimation-and","slug":"balancing-value-underestimation-and","title":"Balancing Value Underestimation and Overestimation with Realistic Actor-Critic","date":"2021-10-19","arxiv_id":"2110.09712","repositories_listed":1,"syntology":null},{"url":"/paper/an-actor-critic-algorithm-with-deep-double","slug":"an-actor-critic-algorithm-with-deep-double","title":"An actor-critic algorithm with policy gradients to solve the job shop scheduling problem using deep double recurrent agents","date":"2021-10-18","arxiv_id":"2110.09076","repositories_listed":1,"syntology":null},{"url":"/paper/in-a-nutshell-the-human-asked-for-this-latent-1","slug":"in-a-nutshell-the-human-asked-for-this-latent-1","title":"In a Nutshell, the Human Asked for This: Latent Goals for Following Temporal Specifications","date":"2021-10-18","arxiv_id":"2110.09461","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/in-a-nutshell-the-human-asked-for-this-latent-1#ran","syntology_url":"https://syntology.ai/paper/2110.09461","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.09461"}},"official":{"repos":["bgleon/latent-goal-architectures"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/next-best-view-estimation-based-on-deep","slug":"next-best-view-estimation-based-on-deep","title":"Next-Best-View Estimation based on Deep Reinforcement Learning for Active Object Classification","date":"2021-10-13","arxiv_id":"2110.06766","repositories_listed":1,"syntology":null},{"url":"/paper/learning-a-subspace-of-policies-for-online-1","slug":"learning-a-subspace-of-policies-for-online-1","title":"Learning a subspace of policies for online adaptation in Reinforcement Learning","date":"2021-10-11","arxiv_id":"2110.05169","repositories_listed":1,"syntology":null},{"url":"/paper/vectorization-of-raster-manga-by-deep","slug":"vectorization-of-raster-manga-by-deep","title":"MARVEL: Raster Manga Vectorization via Primitive-wise Deep Reinforcement Learning","date":"2021-10-10","arxiv_id":"2110.04830","repositories_listed":1,"syntology":null},{"url":"/paper/tikick-toward-playing-multi-agent-football","slug":"tikick-toward-playing-multi-agent-football","title":"TiKick: Towards Playing Multi-agent Football Full Games from Single-agent Demonstrations","date":"2021-10-09","arxiv_id":"2110.04507","repositories_listed":1,"syntology":null},{"url":"/paper/abcp-automatic-block-wise-and-channel-wise","slug":"abcp-automatic-block-wise-and-channel-wise","title":"ABCP: Automatic Block-wise and Channel-wise Network Pruning via Joint Search","date":"2021-10-08","arxiv_id":"2110.03858","repositories_listed":1,"syntology":null},{"url":"/paper/augmenting-reinforcement-learning-with","slug":"augmenting-reinforcement-learning-with","title":"Augmenting Reinforcement Learning with Behavior Primitives for Diverse Manipulation Tasks","date":"2021-10-07","arxiv_id":"2110.03655","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/augmenting-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2110.03655","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.03655"}},"official":{"repos":["UT-Austin-RPL/maple"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/how-to-sense-the-world-leveraging-hierarchy","slug":"how-to-sense-the-world-leveraging-hierarchy","title":"How to Sense the World: Leveraging Hierarchy in Multimodal Perception for Robust Reinforcement Learning Agents","date":"2021-10-07","arxiv_id":"2110.03608","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-solving-the","slug":"deep-reinforcement-learning-for-solving-the","title":"Deep Reinforcement Learning for Solving the Heterogeneous Capacitated Vehicle Routing Problem","date":"2021-10-06","arxiv_id":"2110.02629","repositories_listed":1,"syntology":null},{"url":"/paper/learning-multi-objective-curricula-for-deep","slug":"learning-multi-objective-curricula-for-deep","title":"Learning Multi-Objective Curricula for Robotic Policy Learning","date":"2021-10-06","arxiv_id":"2110.03032","repositories_listed":1,"syntology":null},{"url":"/paper/optimized-recommender-systems-with-deep","slug":"optimized-recommender-systems-with-deep","title":"Optimized Recommender Systems with Deep Reinforcement Learning","date":"2021-10-06","arxiv_id":"2110.03039","repositories_listed":1,"syntology":null},{"url":"/paper/continuous-time-fitted-value-iteration-for","slug":"continuous-time-fitted-value-iteration-for","title":"Continuous-Time Fitted Value Iteration for Robust Policies","date":"2021-10-05","arxiv_id":"2110.01954","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/continuous-time-fitted-value-iteration-for#ran","syntology_url":"https://syntology.ai/paper/2110.01954","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.01954"}},"official":null}},{"url":"/paper/neurwin-neural-whittle-index-network-for-1","slug":"neurwin-neural-whittle-index-network-for-1","title":"NeurWIN: Neural Whittle Index Network For Restless Bandits Via Deep RL","date":"2021-10-05","arxiv_id":"2110.02128","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/neurwin-neural-whittle-index-network-for-1#ran","syntology_url":"https://syntology.ai/paper/2110.02128","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.02128"}},"official":{"repos":["khalednakhleh/NeurWIN"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/collective-explainable-ai-explaining","slug":"collective-explainable-ai-explaining","title":"Collective eXplainable AI: Explaining Cooperative Strategies and Agent Contribution in Multiagent Reinforcement Learning with Shapley Values","date":"2021-10-04","arxiv_id":"2110.01307","repositories_listed":1,"syntology":null},{"url":"/paper/neural-network-verification-in-control","slug":"neural-network-verification-in-control","title":"Neural Network Verification in Control","date":"2021-09-30","arxiv_id":"2110.01388","repositories_listed":1,"syntology":null},{"url":"/paper/unified-data-collection-for-visual-inertial","slug":"unified-data-collection-for-visual-inertial","title":"Unified Data Collection for Visual-Inertial Calibration via Deep Reinforcement Learning","date":"2021-09-30","arxiv_id":"2109.14974","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unified-data-collection-for-visual-inertial#ran","syntology_url":"https://syntology.ai/paper/2109.14974","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.14974"}},"official":{"repos":["ethz-asl/Learn-to-Calibrate"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hyperdqn-a-randomized-exploration-method-for","slug":"hyperdqn-a-randomized-exploration-method-for","title":"HyperDQN: A Randomized Exploration Method for Deep Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-efficient-online-3d-bin-packing-on","slug":"learning-efficient-online-3d-bin-packing-on","title":"Learning Efficient Online 3D Bin Packing on Packing Configuration Trees","date":"2021-09-29","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/minihack-the-planet-a-sandbox-for-open-ended","slug":"minihack-the-planet-a-sandbox-for-open-ended","title":"MiniHack the Planet: A Sandbox for Open-Ended Reinforcement Learning Research","date":"2021-09-27","arxiv_id":"2109.13202","repositories_listed":1,"syntology":null},{"url":"/paper/emergent-behavior-and-neural-dynamics-in","slug":"emergent-behavior-and-neural-dynamics-in","title":"Emergent behavior and neural dynamics in artificial agents tracking turbulent plumes","date":"2021-09-25","arxiv_id":"2109.12434","repositories_listed":1,"syntology":null},{"url":"/paper/parameter-free-deterministic-reduction-of-the","slug":"parameter-free-deterministic-reduction-of-the","title":"Parameter-free Reduction of the Estimation Bias in Deep Reinforcement Learning for Deterministic Policy Gradients","date":"2021-09-24","arxiv_id":"2109.11788","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-navigational-safety-in-crowded","slug":"enhancing-navigational-safety-in-crowded","title":"Enhancing Navigational Safety in Crowded Environments using Semantic-Deep-Reinforcement-Learning-based Navigation","date":"2021-09-23","arxiv_id":"2109.11288","repositories_listed":1,"syntology":null},{"url":"/paper/enero-efficient-real-time-routing","slug":"enero-efficient-real-time-routing","title":"ENERO: Efficient Real-Time WAN Routing Optimization with Deep Reinforcement Learning","date":"2021-09-22","arxiv_id":"2109.10883","repositories_listed":1,"syntology":null},{"url":"/paper/estimation-error-correction-in-deep","slug":"estimation-error-correction-in-deep","title":"Estimation Error Correction in Deep Reinforcement Learning for Deterministic Actor-Critic Methods","date":"2021-09-22","arxiv_id":"2109.10736","repositories_listed":1,"syntology":null},{"url":"/paper/generalization-in-text-based-games-via","slug":"generalization-in-text-based-games-via","title":"Generalization in Text-based Games via Hierarchical Reinforcement Learning","date":"2021-09-21","arxiv_id":"2109.09968","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalization-in-text-based-games-via#ran","syntology_url":"https://syntology.ai/paper/2109.09968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.09968"}},"official":{"repos":["yunqiuxu/h-kga"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchical-policy-for-non-prehensile-multi","slug":"hierarchical-policy-for-non-prehensile-multi","title":"Hierarchical Policy for Non-prehensile Multi-object Rearrangement with Deep Reinforcement Learning and Monte Carlo Tree Search","date":"2021-09-18","arxiv_id":"2109.08973","repositories_listed":1,"syntology":null},{"url":"/paper/targeted-attack-on-deep-rl-based-autonomous","slug":"targeted-attack-on-deep-rl-based-autonomous","title":"Targeted Attack on Deep RL-based Autonomous Driving with Learned Visual Patterns","date":"2021-09-16","arxiv_id":"2109.07723","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/targeted-attack-on-deep-rl-based-autonomous#ran","syntology_url":"https://syntology.ai/paper/2109.07723","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.07723"}},"official":{"repos":["ASU-APG/Targeted-Physical-Adversarial-Attacks-on-AD"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/back-to-basics-deep-reinforcement-learning-in","slug":"back-to-basics-deep-reinforcement-learning-in","title":"Back to Basics: Deep Reinforcement Learning in Traffic Signal Control","date":"2021-09-15","arxiv_id":"2109.07180","repositories_listed":1,"syntology":null},{"url":"/paper/dcur-data-curriculum-for-teaching-via-samples","slug":"dcur-data-curriculum-for-teaching-via-samples","title":"DCUR: Data Curriculum for Teaching via Samples with Reinforcement Learning","date":"2021-09-15","arxiv_id":"2109.07380","repositories_listed":1,"syntology":null},{"url":"/paper/dependability-analysis-of-deep-reinforcement","slug":"dependability-analysis-of-deep-reinforcement","title":"Dependability Analysis of Deep Reinforcement Learning based Robotics and Autonomous Systems through Probabilistic Model Checking","date":"2021-09-14","arxiv_id":"2109.06523","repositories_listed":1,"syntology":null},{"url":"/paper/focus-on-impact-indoor-exploration-with","slug":"focus-on-impact-indoor-exploration-with","title":"Focus on Impact: Indoor Exploration with Intrinsic Motivation","date":"2021-09-14","arxiv_id":"2109.08521","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-navigate-intersections-with","slug":"learning-to-navigate-intersections-with","title":"Learning to Navigate Intersections with Unsupervised Driver Trait Inference","date":"2021-09-14","arxiv_id":"2109.06783","repositories_listed":1,"syntology":null},{"url":"/paper/towards-optimized-actions-in-critical","slug":"towards-optimized-actions-in-critical","title":"Towards optimized actions in critical situations of soccer games with deep reinforcement learning","date":"2021-09-14","arxiv_id":"2109.06625","repositories_listed":1,"syntology":null},{"url":"/paper/wavecorr-correlation-savvy-deep-reinforcement","slug":"wavecorr-correlation-savvy-deep-reinforcement","title":"WaveCorr: Correlation-savvy Deep Reinforcement Learning for Portfolio Management","date":"2021-09-14","arxiv_id":"2109.07005","repositories_listed":1,"syntology":null},{"url":"/paper/direct-random-search-for-fine-tuning-of-deep","slug":"direct-random-search-for-fine-tuning-of-deep","title":"Direct Random Search for Fine Tuning of Deep Reinforcement Learning Policies","date":"2021-09-12","arxiv_id":"2109.05604","repositories_listed":1,"syntology":null},{"url":"/paper/learning-selective-communication-for-multi","slug":"learning-selective-communication-for-multi","title":"Learning Selective Communication for Multi-Agent Path Finding","date":"2021-09-12","arxiv_id":"2109.05413","repositories_listed":1,"syntology":null}],"record_sha256":"49bbe0b7f5f060e61a4a29b90851029eaf77932774c4e905e05398203cbb7cb0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}