{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/12","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":12,"pages_in_order":59,"rows_per_page":100,"rows":[1101,1200],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/11","next":"/task/deep-reinforcement-learning/papers/13","papers":[{"url":"/paper/deep-reinforcement-learning-for-room","slug":"deep-reinforcement-learning-for-room","title":"Deep Reinforcement Learning for room temperature control: a black-box pipeline from data to policies","date":"2021-09-10","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/drop-deep-relocating-option-policy-for","slug":"drop-deep-relocating-option-policy-for","title":"DROP: Deep relocating option policy for optimal ride-hailing vehicle repositioning","date":"2021-09-09","arxiv_id":"2109.04149","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-quantum-variational-circuits-with","slug":"optimizing-quantum-variational-circuits-with","title":"Optimizing Quantum Variational Circuits with Deep Reinforcement Learning","date":"2021-09-07","arxiv_id":"2109.03188","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimizing-quantum-variational-circuits-with#ran","syntology_url":"https://syntology.ai/paper/2109.03188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.03188"}},"official":{"repos":["lockwo/rl_qvc_opt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-exploration-of-deep-learning-methods-in","slug":"an-exploration-of-deep-learning-methods-in","title":"An Exploration of Deep Learning Methods in Hungry Geese","date":"2021-09-05","arxiv_id":"2109.01954","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-object-to-zone-graph-for-object","slug":"hierarchical-object-to-zone-graph-for-object","title":"Hierarchical Object-to-Zone Graph for Object Navigation","date":"2021-09-05","arxiv_id":"2109.02066","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":1,"n_ran_checked":5,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/hierarchical-object-to-zone-graph-for-object#ran","syntology_url":"https://syntology.ai/paper/2109.02066","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.02066"}},"official":{"repos":["sx-zhang/hoz"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":1,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/temporal-aware-deep-reinforcement-learning","slug":"temporal-aware-deep-reinforcement-learning","title":"Temporal Shift Reinforcement Learning","date":"2021-09-05","arxiv_id":"2109.02145","repositories_listed":1,"syntology":null},{"url":"/paper/catastrophic-interference-in-reinforcement","slug":"catastrophic-interference-in-reinforcement","title":"Catastrophic Interference in Reinforcement Learning: A Solution Based on Context Division and Knowledge Distillation","date":"2021-09-01","arxiv_id":"2109.00525","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/catastrophic-interference-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2109.00525","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.00525"}},"official":{"repos":["sweety-dm/interference-aware-deep-q-learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-synthesize-programs-as","slug":"learning-to-synthesize-programs-as","title":"Learning to Synthesize Programs as Interpretable and Generalizable Policies","date":"2021-08-31","arxiv_id":"2108.13643","repositories_listed":1,"syntology":{"n":20,"n_ran":15,"n_constructed":6,"n_ran_checked":8,"n_instrument":7,"n_unverified":5,"n_honours":0,"n_violates":2,"n_no_contract":6,"n_pointer_only":0,"phrase":"15 ran (of which 6 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 2 violated, 6 with no contract checked; 7 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/learning-to-synthesize-programs-as#ran","syntology_url":"https://syntology.ai/paper/2108.13643","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.13643"}},"official":null}},{"url":"/paper/decentralized-autofocusing-system-with","slug":"decentralized-autofocusing-system-with","title":"DASHA: Decentralized Autofocusing System with Hierarchical Agents","date":"2021-08-29","arxiv_id":"2108.12842","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-condition","slug":"reinforcement-learning-based-condition","title":"Reinforcement Learning based Condition-oriented Maintenance Scheduling for Flow Line Systems","date":"2021-08-27","arxiv_id":"2108.12298","repositories_listed":1,"syntology":null},{"url":"/paper/responsive-regulation-of-dynamic-uav","slug":"responsive-regulation-of-dynamic-uav","title":"Responsive Regulation of Dynamic UAV Communication Networks Based on Deep Reinforcement Learning","date":"2021-08-25","arxiv_id":"2108.11012","repositories_listed":1,"syntology":null},{"url":"/paper/diversity-based-trajectory-and-goal-selection","slug":"diversity-based-trajectory-and-goal-selection","title":"Diversity-based Trajectory and Goal Selection with Hindsight Experience Replay","date":"2021-08-17","arxiv_id":"2108.07887","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-robot-navigation","slug":"reinforcement-learning-for-robot-navigation","title":"Reinforcement Learning for Robot Navigation with Adaptive Forward Simulation Time (AFST) in a Semi-Markov Model","date":"2021-08-13","arxiv_id":"2108.06161","repositories_listed":1,"syntology":null},{"url":"/paper/safe-deep-reinforcement-learning-for-multi","slug":"safe-deep-reinforcement-learning-for-multi","title":"Safe Deep Reinforcement Learning for Multi-Agent Systems with Continuous Action Spaces","date":"2021-08-09","arxiv_id":"2108.03952","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safe-deep-reinforcement-learning-for-multi#ran","syntology_url":"https://syntology.ai/paper/2108.03952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.03952"}},"official":{"repos":["zisikons/deep-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-ai-economist-optimal-economic-policy","slug":"the-ai-economist-optimal-economic-policy","title":"The AI Economist: Optimal Economic Policy Design via Two-level Deep Reinforcement Learning","date":"2021-08-05","arxiv_id":"2108.02755","repositories_listed":1,"syntology":null},{"url":"/paper/risk-conditioned-neural-motion-planning","slug":"risk-conditioned-neural-motion-planning","title":"Risk Conditioned Neural Motion Planning","date":"2021-08-04","arxiv_id":"2108.01851","repositories_listed":1,"syntology":null},{"url":"/paper/tianshou-a-highly-modularized-deep","slug":"tianshou-a-highly-modularized-deep","title":"Tianshou: a Highly Modularized Deep Reinforcement Learning Library","date":"2021-07-29","arxiv_id":"2107.14171","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tianshou-a-highly-modularized-deep#ran","syntology_url":"https://syntology.ai/paper/2107.14171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.14171"}},"official":{"repos":["thu-ml/tianshou"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/finding-failures-in-high-fidelity-simulation","slug":"finding-failures-in-high-fidelity-simulation","title":"Finding Failures in High-Fidelity Simulation using Adaptive Stress Testing and the Backward Algorithm","date":"2021-07-27","arxiv_id":"2107.12940","repositories_listed":1,"syntology":null},{"url":"/paper/hindsight-value-function-for-variance","slug":"hindsight-value-function-for-variance","title":"Hindsight Value Function for Variance Reduction in Stochastic Dynamic Environment","date":"2021-07-26","arxiv_id":"2107.12216","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":6,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 6 samples that ran constructed an object rather than computing a result","sample_list":"/paper/hindsight-value-function-for-variance#ran","syntology_url":"https://syntology.ai/paper/2107.12216","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.12216"}},"official":null}},{"url":"/paper/co-designing-intelligent-control-of-building","slug":"co-designing-intelligent-control-of-building","title":"Co-designing Intelligent Control of Building HVACs and Microgrids","date":"2021-07-18","arxiv_id":"2107.08378","repositories_listed":1,"syntology":null},{"url":"/paper/rellie-deep-reinforcement-learning-for","slug":"rellie-deep-reinforcement-learning-for","title":"ReLLIE: Deep Reinforcement Learning for Customized Low-Light Image Enhancement","date":"2021-07-13","arxiv_id":"2107.05830","repositories_listed":1,"syntology":null},{"url":"/paper/learning-interaction-aware-guidance-policies","slug":"learning-interaction-aware-guidance-policies","title":"Learning Interaction-aware Guidance Policies for Motion Planning in Dense Traffic Scenarios","date":"2021-07-09","arxiv_id":"2107.04538","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-online-service-coordination-using","slug":"distributed-online-service-coordination-using","title":"Distributed Online Service Coordination Using Deep Reinforcement Learning","date":"2021-07-07","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-modal-mutual-information-mummi-training","slug":"multi-modal-mutual-information-mummi-training","title":"Multi-Modal Mutual Information (MuMMI) Training for Robust Self-Supervised Deep Reinforcement Learning","date":"2021-07-06","arxiv_id":"2107.02339","repositories_listed":1,"syntology":null},{"url":"/paper/ensemble-and-auxiliary-tasks-for-data","slug":"ensemble-and-auxiliary-tasks-for-data","title":"Ensemble and Auxiliary Tasks for Data-Efficient Deep Reinforcement Learning","date":"2021-07-05","arxiv_id":"2107.01904","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ensemble-and-auxiliary-tasks-for-data#ran","syntology_url":"https://syntology.ai/paper/2107.01904","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.01904"}},"official":{"repos":["NUS-LID/RENAULT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-efficient-reinforcement-learning-via-2","slug":"sample-efficient-reinforcement-learning-via-2","title":"Sample Efficient Reinforcement Learning via Model-Ensemble Exploration and Exploitation","date":"2021-07-05","arxiv_id":"2107.01825","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-adversarial-attacks-on-1","slug":"understanding-adversarial-attacks-on-1","title":"Understanding Adversarial Attacks on Observations in Deep Reinforcement Learning","date":"2021-06-30","arxiv_id":"2106.15860","repositories_listed":1,"syntology":null},{"url":"/paper/a-nonlinear-hidden-layer-enables-actor-critic","slug":"a-nonlinear-hidden-layer-enables-actor-critic","title":"A nonlinear hidden layer enables actor-critic agents to learn multiple paired association navigation","date":"2021-06-25","arxiv_id":"2106.13541","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-phy-layer","slug":"reinforcement-learning-for-phy-layer","title":"Reinforcement Learning for Physical Layer Communications","date":"2021-06-22","arxiv_id":"2106.11595","repositories_listed":1,"syntology":null},{"url":"/paper/predicting-human-scanpaths-in-visual-question","slug":"predicting-human-scanpaths-in-visual-question","title":"Predicting Human Scanpaths in Visual Question Answering","date":"2021-06-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/towards-safe-reinforcement-learning-via-1","slug":"towards-safe-reinforcement-learning-via-1","title":"Towards Safe Reinforcement Learning via Constraining Conditional Value at Risk","date":"2021-06-18","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/real-time-attacks-against-deep-reinforcement","slug":"real-time-attacks-against-deep-reinforcement","title":"Real-time Adversarial Perturbations against Deep Reinforcement Learning Policies: Attacks and Defenses","date":"2021-06-16","arxiv_id":"2106.08746","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/real-time-attacks-against-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2106.08746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.08746"}},"official":{"repos":["ssg-research/ad3-action-distribution-divergence-detector"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-conservation","slug":"deep-reinforcement-learning-for-conservation","title":"Deep Reinforcement Learning for Conservation Decisions","date":"2021-06-15","arxiv_id":"2106.08272","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based-group","slug":"deep-reinforcement-learning-based-group","title":"Deep Reinforcement Learning based Group Recommender System","date":"2021-06-13","arxiv_id":"2106.06900","repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-reinforcement-learning-approach-to-4","slug":"a-deep-reinforcement-learning-approach-to-4","title":"A Deep Reinforcement Learning Approach to Marginalized Importance Sampling with the Successor Representation","date":"2021-06-12","arxiv_id":"2106.06854","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/a-deep-reinforcement-learning-approach-to-4#ran","syntology_url":"https://syntology.ai/paper/2106.06854","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.06854"}},"official":{"repos":["sfujim/SR-DICE"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/douzero-mastering-doudizhu-with-self-play","slug":"douzero-mastering-doudizhu-with-self-play","title":"DouZero: Mastering DouDizhu with Self-Play Deep Reinforcement Learning","date":"2021-06-11","arxiv_id":"2106.06135","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/douzero-mastering-doudizhu-with-self-play#ran","syntology_url":"https://syntology.ai/paper/2106.06135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.06135"}},"official":{"repos":["kwai/DouZero"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"url":"/paper/achieving-diverse-objectives-with-ai-driven","slug":"achieving-diverse-objectives-with-ai-driven","title":"AI-driven Prices for Externalities and Sustainability in Production Markets","date":"2021-06-10","arxiv_id":"2106.06060","repositories_listed":1,"syntology":null},{"url":"/paper/simplifying-deep-reinforcement-learning-via","slug":"simplifying-deep-reinforcement-learning-via","title":"Simplifying Deep Reinforcement Learning via Self-Supervision","date":"2021-06-10","arxiv_id":"2106.05526","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/simplifying-deep-reinforcement-learning-via#ran","syntology_url":"https://syntology.ai/paper/2106.05526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.05526"}},"official":{"repos":["daochenzha/SSRL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pretrained-encoders-are-all-you-need","slug":"pretrained-encoders-are-all-you-need","title":"Pretrained Encoders are All You Need","date":"2021-06-09","arxiv_id":"2106.05139","repositories_listed":1,"syntology":null},{"url":"/paper/pretraining-representations-for-data","slug":"pretraining-representations-for-data","title":"Pretraining Representations for Data-Efficient Reinforcement Learning","date":"2021-06-09","arxiv_id":"2106.04799","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pretraining-representations-for-data#ran","syntology_url":"https://syntology.ai/paper/2106.04799","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04799"}},"official":{"repos":["mila-iqia/SGI"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-sparse-training-for-deep","slug":"dynamic-sparse-training-for-deep","title":"Dynamic Sparse Training for Deep Reinforcement Learning","date":"2021-06-08","arxiv_id":"2106.04217","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-sparse-training-for-deep#ran","syntology_url":"https://syntology.ai/paper/2106.04217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04217"}},"official":{"repos":["GhadaSokar/Dynamic-Sparse-Training-for-Deep-Reinforcement-Learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-markov-state-abstractions-for-deep","slug":"learning-markov-state-abstractions-for-deep","title":"Learning Markov State Abstractions for Deep Reinforcement Learning","date":"2021-06-08","arxiv_id":"2106.04379","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-markov-state-abstractions-for-deep#ran","syntology_url":"https://syntology.ai/paper/2106.04379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04379"}},"official":{"repos":["camall3n/markov-state-abstractions"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/correcting-momentum-in-temporal-difference-1","slug":"correcting-momentum-in-temporal-difference-1","title":"Correcting Momentum in Temporal Difference Learning","date":"2021-06-07","arxiv_id":"2106.03955","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/correcting-momentum-in-temporal-difference-1#ran","syntology_url":"https://syntology.ai/paper/2106.03955","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03955"}},"official":{"repos":["bengioe/staleness-corrected-momentum"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-target-networks-improving-deep-q","slug":"beyond-target-networks-improving-deep-q","title":"Bridging the Gap Between Target Networks and Functional Regularization","date":"2021-06-04","arxiv_id":"2106.02613","repositories_listed":1,"syntology":null},{"url":"/paper/model-agnostic-and-scalable-counterfactual","slug":"model-agnostic-and-scalable-counterfactual","title":"Model-agnostic and Scalable Counterfactual Explanations via Reinforcement Learning","date":"2021-06-04","arxiv_id":"2106.02597","repositories_listed":1,"syntology":null},{"url":"/paper/a-consciousness-inspired-planning-agent-for","slug":"a-consciousness-inspired-planning-agent-for","title":"A Consciousness-Inspired Planning Agent for Model-Based Reinforcement Learning","date":"2021-06-03","arxiv_id":"2106.02097","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":3,"n_ran_checked":3,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/a-consciousness-inspired-planning-agent-for#ran","syntology_url":"https://syntology.ai/paper/2106.02097","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.02097"}},"official":{"repos":["mila-iqia/conscious-planning"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/high-quality-diversification-for-task","slug":"high-quality-diversification-for-task","title":"High-Quality Diversification for Task-Oriented Dialogue Systems","date":"2021-06-02","arxiv_id":"2106.00891","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-in-quantitative","slug":"deep-reinforcement-learning-in-quantitative","title":"Deep Reinforcement Learning in Quantitative Algorithmic Trading: A Review","date":"2021-05-31","arxiv_id":"2106.00123","repositories_listed":1,"syntology":null},{"url":"/paper/robust-value-iteration-for-continuous-control","slug":"robust-value-iteration-for-continuous-control","title":"Robust Value Iteration for Continuous Control Tasks","date":"2021-05-25","arxiv_id":"2105.12189","repositories_listed":1,"syntology":null},{"url":"/paper/towards-scalable-verification-of-rl-driven","slug":"towards-scalable-verification-of-rl-driven","title":"Towards Scalable Verification of Deep Reinforcement Learning","date":"2021-05-25","arxiv_id":"2105.11931","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-optimal-3","slug":"deep-reinforcement-learning-for-optimal-3","title":"Deep Reinforcement Learning for Optimal Stopping with Application in Financial Engineering","date":"2021-05-19","arxiv_id":"2105.08877","repositories_listed":1,"syntology":null},{"url":"/paper/principled-exploration-via-optimistic","slug":"principled-exploration-via-optimistic","title":"Principled Exploration via Optimistic Bootstrapping and Backward Induction","date":"2021-05-13","arxiv_id":"2105.06022","repositories_listed":1,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":12,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/principled-exploration-via-optimistic#ran","syntology_url":"https://syntology.ai/paper/2105.06022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.06022"}},"official":{"repos":["Baichenjia/OB2I"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/spectral-normalisation-for-deep-reinforcement","slug":"spectral-normalisation-for-deep-reinforcement","title":"Spectral Normalisation for Deep Reinforcement Learning: an Optimisation Perspective","date":"2021-05-11","arxiv_id":"2105.05246","repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-reinforcement-learning-approach-to-3","slug":"a-deep-reinforcement-learning-approach-to-3","title":"A Deep Reinforcement Learning Approach to Audio-Based Navigation in a Multi-Speaker Environment","date":"2021-05-10","arxiv_id":"2105.04488","repositories_listed":1,"syntology":null},{"url":"/paper/value-iteration-in-continuous-actions-states","slug":"value-iteration-in-continuous-actions-states","title":"Value Iteration in Continuous Actions, States and Time","date":"2021-05-10","arxiv_id":"2105.04682","repositories_listed":1,"syntology":null},{"url":"/paper/generative-actor-critic-an-off-policy","slug":"generative-actor-critic-an-off-policy","title":"Generative Actor-Critic: An Off-policy Algorithm Using the Push-forward Model","date":"2021-05-08","arxiv_id":"2105.03733","repositories_listed":1,"syntology":null},{"url":"/paper/context-based-soft-actor-critic-for","slug":"context-based-soft-actor-critic-for","title":"Context-Based Soft Actor Critic for Environments with Non-stationary Dynamics","date":"2021-05-07","arxiv_id":"2105.03310","repositories_listed":1,"syntology":null},{"url":"/paper/deeprf-deep-reinforcement-learning-designed","slug":"deeprf-deep-reinforcement-learning-designed","title":"Deep reinforcement learning-designed radiofrequency waveform in MRI","date":"2021-05-07","arxiv_id":"2105.03061","repositories_listed":1,"syntology":null},{"url":"/paper/meta-learning-based-deep-reinforcement","slug":"meta-learning-based-deep-reinforcement","title":"Meta-Learning-Based Deep Reinforcement Learning for Multiobjective Optimization Problems","date":"2021-05-06","arxiv_id":"2105.02741","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/meta-learning-based-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2105.02741","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.02741"}},"official":{"repos":["zhangzizhen/ml-dam"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/solve-routing-problems-with-a-residual-edge","slug":"solve-routing-problems-with-a-residual-edge","title":"Solve routing problems with a residual edge-graph attention neural network","date":"2021-05-06","arxiv_id":"2105.02730","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-adaptive-3","slug":"deep-reinforcement-learning-for-adaptive-3","title":"Deep Reinforcement Learning for Adaptive Exploration of Unknown Environments","date":"2021-05-04","arxiv_id":"2105.01606","repositories_listed":1,"syntology":null},{"url":"/paper/curious-exploration-and-return-based-memory","slug":"curious-exploration-and-return-based-memory","title":"Curious Exploration and Return-based Memory Restoration for Deep Reinforcement Learning","date":"2021-05-02","arxiv_id":"2105.00499","repositories_listed":1,"syntology":null},{"url":"/paper/adapting-to-reward-progressivity-via-spectral-1","slug":"adapting-to-reward-progressivity-via-spectral-1","title":"Adapting to Reward Progressivity via Spectral Reinforcement Learning","date":"2021-04-29","arxiv_id":"2104.14138","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/adapting-to-reward-progressivity-via-spectral-1#ran","syntology_url":"https://syntology.ai/paper/2104.14138","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.14138"}},"official":{"repos":["mchldann/SpectralDQN"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-trading-with-predictable","slug":"deep-reinforcement-trading-with-predictable","title":"Deep Reinforcement Trading with Predictable Returns","date":"2021-04-29","arxiv_id":"2104.14683","repositories_listed":1,"syntology":null},{"url":"/paper/a-scalable-and-reproducible-system-on-chip","slug":"a-scalable-and-reproducible-system-on-chip","title":"A Scalable and Reproducible System-on-Chip Simulation for Reinforcement Learning","date":"2021-04-27","arxiv_id":"2104.13187","repositories_listed":1,"syntology":null},{"url":"/paper/computational-performance-of-deep","slug":"computational-performance-of-deep","title":"Computational Performance of Deep Reinforcement Learning to find Nash Equilibria","date":"2021-04-26","arxiv_id":"2104.12895","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-grasping-policies-for-human-in-the","slug":"end-to-end-grasping-policies-for-human-in-the","title":"End-to-end grasping policies for human-in-the-loop robots via deep reinforcement learning","date":"2021-04-26","arxiv_id":"2104.12842","repositories_listed":1,"syntology":null},{"url":"/paper/graph-neural-network-reinforcement-learning","slug":"graph-neural-network-reinforcement-learning","title":"Graph Neural Network Reinforcement Learning for Autonomous Mobility-on-Demand Systems","date":"2021-04-23","arxiv_id":"2104.11434","repositories_listed":1,"syntology":null},{"url":"/paper/xai-n-sensor-based-robot-navigation-using","slug":"xai-n-sensor-based-robot-navigation-using","title":"XAI-N: Sensor-based Robot Navigation using Expert Policies and Decision Trees","date":"2021-04-22","arxiv_id":"2104.10818","repositories_listed":1,"syntology":null},{"url":"/paper/probabilistic-mixture-of-experts-for-1","slug":"probabilistic-mixture-of-experts-for-1","title":"Probabilistic Mixture-of-Experts for Efficient Deep Reinforcement Learning","date":"2021-04-19","arxiv_id":"2104.09122","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/probabilistic-mixture-of-experts-for-1#ran","syntology_url":"https://syntology.ai/paper/2104.09122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.09122"}},"official":{"repos":["JieRen98/rlkit-pmoe"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/quick-learner-automated-vehicle-adapting-its","slug":"quick-learner-automated-vehicle-adapting-its","title":"Quick Learner Automated Vehicle Adapting its Roadmanship to Varying Traffic Cultures with Meta Reinforcement Learning","date":"2021-04-18","arxiv_id":"2104.08876","repositories_listed":1,"syntology":null},{"url":"/paper/learning-on-a-budget-via-teacher-imitation","slug":"learning-on-a-budget-via-teacher-imitation","title":"Learning on a Budget via Teacher Imitation","date":"2021-04-17","arxiv_id":"2104.08440","repositories_listed":1,"syntology":null},{"url":"/paper/quantum-architecture-search-via-deep","slug":"quantum-architecture-search-via-deep","title":"Quantum Architecture Search via Deep Reinforcement Learning","date":"2021-04-15","arxiv_id":"2104.07715","repositories_listed":1,"syntology":null},{"url":"/paper/decomposed-soft-actor-critic-method-for","slug":"decomposed-soft-actor-critic-method-for","title":"Decomposed Soft Actor-Critic Method for Cooperative Multi-Agent Reinforcement Learning","date":"2021-04-14","arxiv_id":"2104.06655","repositories_listed":1,"syntology":null},{"url":"/paper/the-atari-data-scraper","slug":"the-atari-data-scraper","title":"The Atari Data Scraper","date":"2021-04-11","arxiv_id":"2104.04893","repositories_listed":1,"syntology":null},{"url":"/paper/behavior-guided-actor-critic-improving","slug":"behavior-guided-actor-critic-improving","title":"Behavior-Guided Actor-Critic: Improving Exploration via Learning Policy Behavior Representation for Deep Reinforcement Learning","date":"2021-04-09","arxiv_id":"2104.04424","repositories_listed":1,"syntology":null},{"url":"/paper/connecting-deep-reinforcement-learning-based","slug":"connecting-deep-reinforcement-learning-based","title":"Connecting Deep-Reinforcement-Learning-based Obstacle Avoidance with Conventional Global Planners using Waypoint Generators","date":"2021-04-08","arxiv_id":"2104.03663","repositories_listed":1,"syntology":null},{"url":"/paper/towards-deployment-of-deep-reinforcement","slug":"towards-deployment-of-deep-reinforcement","title":"Arena-Rosnav: Towards Deployment of Deep-Reinforcement-Learning-Based Obstacle Avoidance into Conventional Autonomous Navigation Systems","date":"2021-04-08","arxiv_id":"2104.03616","repositories_listed":1,"syntology":null},{"url":"/paper/influencing-reinforcement-learning-through","slug":"influencing-reinforcement-learning-through","title":"Influencing Reinforcement Learning through Natural Language Guidance","date":"2021-04-04","arxiv_id":"2104.01506","repositories_listed":1,"syntology":null},{"url":"/paper/identifying-co-adaptation-of-algorithmic-and","slug":"identifying-co-adaptation-of-algorithmic-and","title":"Co-Adaptation of Algorithmic and Implementational Innovations in Inference-based Deep Reinforcement Learning","date":"2021-03-31","arxiv_id":"2103.17258","repositories_listed":1,"syntology":null},{"url":"/paper/simultaneous-navigation-and-construction","slug":"simultaneous-navigation-and-construction","title":"Simultaneous Navigation and Construction Benchmarking Environments","date":"2021-03-31","arxiv_id":"2103.16732","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-resource-1","slug":"deep-reinforcement-learning-for-resource-1","title":"Deep Reinforcement Learning for Resource Allocation in Business Processes","date":"2021-03-29","arxiv_id":"2104.00541","repositories_listed":1,"syntology":null},{"url":"/paper/agent-with-warm-start-and-adaptive-dynamic","slug":"agent-with-warm-start-and-adaptive-dynamic","title":"Agent with Warm Start and Adaptive Dynamic Termination for Plane Localization in 3D Ultrasound","date":"2021-03-26","arxiv_id":"2103.14502","repositories_listed":1,"syntology":null},{"url":"/paper/character-controllers-using-motion-vaes","slug":"character-controllers-using-motion-vaes","title":"Character Controllers Using Motion VAEs","date":"2021-03-26","arxiv_id":"2103.14274","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/character-controllers-using-motion-vaes#ran","syntology_url":"https://syntology.ai/paper/2103.14274","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.14274"}},"official":{"repos":["electronicarts/character-motion-vaes"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/medselect-selective-labeling-for-medical","slug":"medselect-selective-labeling-for-medical","title":"MedSelect: Selective Labeling for Medical Image Classification Combining Meta-Learning with Deep Reinforcement Learning","date":"2021-03-26","arxiv_id":"2103.14339","repositories_listed":1,"syntology":null},{"url":"/paper/omnihang-learning-to-hang-arbitrary-objects","slug":"omnihang-learning-to-hang-arbitrary-objects","title":"OmniHang: Learning to Hang Arbitrary Objects using Contact Point Correspondences and Neural Collision Estimation","date":"2021-03-26","arxiv_id":"2103.14283","repositories_listed":1,"syntology":null},{"url":"/paper/model-predictive-actor-critic-accelerating","slug":"model-predictive-actor-critic-accelerating","title":"Model Predictive Actor-Critic: Accelerating Robot Skill Acquisition with Deep Reinforcement Learning","date":"2021-03-25","arxiv_id":"2103.13842","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-predictive-actor-critic-accelerating#ran","syntology_url":"https://syntology.ai/paper/2103.13842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.13842"}},"official":{"repos":["dnandha/mopac"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-information-capacity-information","slug":"policy-information-capacity-information","title":"Policy Information Capacity: Information-Theoretic Measure for Task Complexity in Deep Reinforcement Learning","date":"2021-03-23","arxiv_id":"2103.12726","repositories_listed":1,"syntology":null},{"url":"/paper/spatial-intention-maps-for-multi-agent-mobile","slug":"spatial-intention-maps-for-multi-agent-mobile","title":"Spatial Intention Maps for Multi-Agent Mobile Manipulation","date":"2021-03-23","arxiv_id":"2103.12710","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/spatial-intention-maps-for-multi-agent-mobile#ran","syntology_url":"https://syntology.ai/paper/2103.12710","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.12710"}},"official":{"repos":["jimmyyhwu/spatial-intention-maps"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-recommend-frame-for-interactive","slug":"learning-to-recommend-frame-for-interactive","title":"Learning to Recommend Frame for Interactive Video Object Segmentation in the Wild","date":"2021-03-18","arxiv_id":"2103.10391","repositories_listed":1,"syntology":null},{"url":"/paper/teachmyagent-a-benchmark-for-automatic","slug":"teachmyagent-a-benchmark-for-automatic","title":"TeachMyAgent: a Benchmark for Automatic Curriculum Learning in Deep RL","date":"2021-03-17","arxiv_id":"2103.09815","repositories_listed":1,"syntology":null},{"url":"/paper/building-safer-autonomous-agents-by","slug":"building-safer-autonomous-agents-by","title":"Building Safer Autonomous Agents by Leveraging Risky Driving Behavior Knowledge","date":"2021-03-16","arxiv_id":"2103.10245","repositories_listed":1,"syntology":null},{"url":"/paper/inclined-quadrotor-landing-using-deep","slug":"inclined-quadrotor-landing-using-deep","title":"Inclined Quadrotor Landing using Deep Reinforcement Learning","date":"2021-03-16","arxiv_id":"2103.09043","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-band","slug":"deep-reinforcement-learning-for-band","title":"Deep Reinforcement Learning for Band Selection in Hyperspectral Image Classification","date":"2021-03-15","arxiv_id":"2103.08741","repositories_listed":1,"syntology":null},{"url":"/paper/large-batch-simulation-for-deep-reinforcement-1","slug":"large-batch-simulation-for-deep-reinforcement-1","title":"Large Batch Simulation for Deep Reinforcement Learning","date":"2021-03-12","arxiv_id":"2103.07013","repositories_listed":1,"syntology":null},{"url":"/paper/generalizable-episodic-memory-for-deep","slug":"generalizable-episodic-memory-for-deep","title":"Generalizable Episodic Memory for Deep Reinforcement Learning","date":"2021-03-11","arxiv_id":"2103.06469","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalizable-episodic-memory-for-deep#ran","syntology_url":"https://syntology.ai/paper/2103.06469","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.06469"}},"official":{"repos":["MouseHu/GEM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/iterative-shrinking-for-referring-expression","slug":"iterative-shrinking-for-referring-expression","title":"Iterative Shrinking for Referring Expression Grounding Using Deep Reinforcement Learning","date":"2021-03-09","arxiv_id":"2103.05187","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-play-soccer-from-scratch-sample","slug":"learning-to-play-soccer-from-scratch-sample","title":"Learning to Play Soccer From Scratch: Sample-Efficient Emergent Coordination through Curriculum-Learning and Competition","date":"2021-03-09","arxiv_id":"2103.05174","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-versus-model-free-deep","slug":"model-based-versus-model-free-deep","title":"Latent Imagination Facilitates Zero-Shot Transfer in Autonomous Racing","date":"2021-03-08","arxiv_id":"2103.04909","repositories_listed":1,"syntology":null},{"url":"/paper/ganav-group-wise-attention-network-for","slug":"ganav-group-wise-attention-network-for","title":"GANav: Efficient Terrain Segmentation for Robot Navigation in Unstructured Outdoor Environments","date":"2021-03-07","arxiv_id":"2103.04233","repositories_listed":1,"syntology":null}],"record_sha256":"fada10be0acb7610414ebc70a3e44980844dde9bff81d0fc89f1b50e6d879d0d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}