{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/4","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":59,"rows_per_page":100,"rows":[301,400],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/3","next":"/task/deep-reinforcement-learning/papers/5","papers":[{"url":"/paper/tstarbots-defeating-the-cheating-level","slug":"tstarbots-defeating-the-cheating-level","title":"TStarBots: Defeating the Cheating Level Builtin AI in StarCraft II in the Full Game","date":"2018-09-19","arxiv_id":"1809.07193","repositories_listed":2,"syntology":null},{"url":"/paper/multi-task-deep-reinforcement-learning-with","slug":"multi-task-deep-reinforcement-learning-with","title":"Multi-task Deep Reinforcement Learning with PopArt","date":"2018-09-12","arxiv_id":"1809.04474","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-task-deep-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/1809.04474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.04474"}},"official":null}},{"url":"/paper/remember-and-forget-for-experience-replay","slug":"remember-and-forget-for-experience-replay","title":"Remember and Forget for Experience Replay","date":"2018-07-16","arxiv_id":"1807.05827","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/remember-and-forget-for-experience-replay#ran","syntology_url":"https://syntology.ai/paper/1807.05827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.05827"}},"official":{"repos":["cselab/smarties"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/algorithmic-framework-for-model-based-deep","slug":"algorithmic-framework-for-model-based-deep","title":"Algorithmic Framework for Model-based Deep Reinforcement Learning with Theoretical Guarantees","date":"2018-07-10","arxiv_id":"1807.03858","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/algorithmic-framework-for-model-based-deep#ran","syntology_url":"https://syntology.ai/paper/1807.03858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.03858"}},"official":{"repos":["roosephu/slbo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/accuracy-based-curriculum-learning-in-deep","slug":"accuracy-based-curriculum-learning-in-deep","title":"Accuracy-based Curriculum Learning in Deep Reinforcement Learning","date":"2018-06-25","arxiv_id":"1806.09614","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-general-video","slug":"deep-reinforcement-learning-for-general-video","title":"Deep Reinforcement Learning for General Video Game AI","date":"2018-06-06","arxiv_id":"1806.02448","repositories_listed":2,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-for-general-video#ran","syntology_url":"https://syntology.ai/paper/1806.02448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.02448"}},"official":{"repos":["rubenrtorrado/GVGAI_GYM"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/robust-distant-supervision-relation","slug":"robust-distant-supervision-relation","title":"Robust Distant Supervision Relation Extraction via Deep Reinforcement Learning","date":"2018-05-24","arxiv_id":"1805.09927","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-distant-supervision-relation#ran","syntology_url":"https://syntology.ai/paper/1805.09927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.09927"}},"official":null}},{"url":"/paper/verifiable-reinforcement-learning-via-policy","slug":"verifiable-reinforcement-learning-via-policy","title":"Verifiable Reinforcement Learning via Policy Extraction","date":"2018-05-22","arxiv_id":"1805.08328","repositories_listed":2,"syntology":null},{"url":"/paper/crafting-a-toolchain-for-image-restoration-by","slug":"crafting-a-toolchain-for-image-restoration-by","title":"Crafting a Toolchain for Image Restoration by Deep Reinforcement Learning","date":"2018-04-10","arxiv_id":"1804.03312","repositories_listed":2,"syntology":null},{"url":"/paper/learning-to-run-challenge-solutions-adapting","slug":"learning-to-run-challenge-solutions-adapting","title":"Learning to Run challenge solutions: Adapting reinforcement learning methods for neuromusculoskeletal environments","date":"2018-04-02","arxiv_id":"1804.00361","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-to-run-challenge-solutions-adapting#ran","syntology_url":"https://syntology.ai/paper/1804.00361","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.00361"}},"official":{"repos":["AdamStelmaszczyk/learning2run"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-time-series","slug":"deep-reinforcement-learning-for-time-series","title":"Deep reinforcement learning for time series: playing idealized trading games","date":"2018-03-11","arxiv_id":"1803.03916","repositories_listed":2,"syntology":null},{"url":"/paper/using-deep-q-learning-to-understand-the-tax","slug":"using-deep-q-learning-to-understand-the-tax","title":"Using deep Q-learning to understand the tax evasion behavior of risk-averse firms","date":"2018-01-29","arxiv_id":"1801.09466","repositories_listed":2,"syntology":null},{"url":"/paper/learning-symmetric-and-low-energy-locomotion","slug":"learning-symmetric-and-low-energy-locomotion","title":"Learning Symmetric and Low-energy Locomotion","date":"2018-01-24","arxiv_id":"1801.08093","repositories_listed":2,"syntology":null},{"url":"/paper/improving-exploration-in-evolution-strategies","slug":"improving-exploration-in-evolution-strategies","title":"Improving Exploration in Evolution Strategies for Deep Reinforcement Learning via a Population of Novelty-Seeking Agents","date":"2017-12-18","arxiv_id":"1712.06560","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-exploration-in-evolution-strategies#ran","syntology_url":"https://syntology.ai/paper/1712.06560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1712.06560"}},"official":{"repos":["uber-research/deep-neuroevolution"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/ai2-thor-an-interactive-3d-environment-for","slug":"ai2-thor-an-interactive-3d-environment-for","title":"AI2-THOR: An Interactive 3D Environment for Visual AI","date":"2017-12-14","arxiv_id":"1712.05474","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ai2-thor-an-interactive-3d-environment-for#ran","syntology_url":"https://syntology.ai/paper/1712.05474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1712.05474"}},"official":{"repos":["allenai/ai2thor"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/minos-multimodal-indoor-simulator-for","slug":"minos-multimodal-indoor-simulator-for","title":"MINOS: Multimodal Indoor Simulator for Navigation in Complex Environments","date":"2017-12-11","arxiv_id":"1712.03931","repositories_listed":2,"syntology":null},{"url":"/paper/ai-safety-gridworlds","slug":"ai-safety-gridworlds","title":"AI Safety Gridworlds","date":"2017-11-27","arxiv_id":"1711.09883","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-sepsis","slug":"deep-reinforcement-learning-for-sepsis","title":"Deep Reinforcement Learning for Sepsis Treatment","date":"2017-11-27","arxiv_id":"1711.09602","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-for-sepsis#ran","syntology_url":"https://syntology.ai/paper/1711.09602","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1711.09602"}},"official":{"repos":["darkefyre/sepsisrl"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/detecting-adversarial-attacks-on-neural","slug":"detecting-adversarial-attacks-on-neural","title":"Detecting Adversarial Attacks on Neural Network Policies with Visual Foresight","date":"2017-10-02","arxiv_id":"1710.00814","repositories_listed":2,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/detecting-adversarial-attacks-on-neural#ran","syntology_url":"https://syntology.ai/paper/1710.00814","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.00814"}},"official":{"repos":["yenchenlin/rl-attack-detection"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/self-supervised-deep-reinforcement-learning","slug":"self-supervised-deep-reinforcement-learning","title":"Self-supervised Deep Reinforcement Learning with Generalized Computation Graphs for Robot Navigation","date":"2017-09-29","arxiv_id":"1709.10489","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-supervised-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1709.10489","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1709.10489"}},"official":{"repos":["gkahn13/gcg"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/deep-tamer-interactive-agent-shaping-in-high","slug":"deep-tamer-interactive-agent-shaping-in-high","title":"Deep TAMER: Interactive Agent Shaping in High-Dimensional State Spaces","date":"2017-09-28","arxiv_id":"1709.10163","repositories_listed":2,"syntology":null},{"url":"/paper/towards-optimally-decentralized-multi-robot","slug":"towards-optimally-decentralized-multi-robot","title":"Towards Optimally Decentralized Multi-Robot Collision Avoidance via Deep Reinforcement Learning","date":"2017-09-28","arxiv_id":"1709.10082","repositories_listed":2,"syntology":null},{"url":"/paper/imagination-augmented-agents-for-deep","slug":"imagination-augmented-agents-for-deep","title":"Imagination-Augmented Agents for Deep Reinforcement Learning","date":"2017-07-19","arxiv_id":"1707.06203","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/imagination-augmented-agents-for-deep#ran","syntology_url":"https://syntology.ai/paper/1707.06203","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1707.06203"}},"official":null}},{"url":"/paper/value-prediction-network","slug":"value-prediction-network","title":"Value Prediction Network","date":"2017-07-11","arxiv_id":"1707.03497","repositories_listed":2,"syntology":null},{"url":"/paper/stochastic-neural-networks-for-hierarchical","slug":"stochastic-neural-networks-for-hierarchical","title":"Stochastic Neural Networks for Hierarchical Reinforcement Learning","date":"2017-04-10","arxiv_id":"1704.03012","repositories_listed":2,"syntology":null},{"url":"/paper/learning-visual-servoing-with-deep-features","slug":"learning-visual-servoing-with-deep-features","title":"Learning Visual Servoing with Deep Features and Fitted Q-Iteration","date":"2017-03-31","arxiv_id":"1703.11000","repositories_listed":2,"syntology":null},{"url":"/paper/socially-aware-motion-planning-with-deep","slug":"socially-aware-motion-planning-with-deep","title":"Socially Aware Motion Planning with Deep Reinforcement Learning","date":"2017-03-26","arxiv_id":"1703.08862","repositories_listed":2,"syntology":null},{"url":"/paper/end-to-end-optimization-of-goal-driven-and","slug":"end-to-end-optimization-of-goal-driven-and","title":"End-to-end optimization of goal-driven and visually grounded dialogue systems","date":"2017-03-15","arxiv_id":"1703.05423","repositories_listed":2,"syntology":null},{"url":"/paper/autonomous-braking-system-via-deep","slug":"autonomous-braking-system-via-deep","title":"Autonomous Braking System via Deep Reinforcement Learning","date":"2017-02-08","arxiv_id":"1702.02302","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-an-overview","slug":"deep-reinforcement-learning-an-overview","title":"Deep Reinforcement Learning: An Overview","date":"2017-01-25","arxiv_id":"1701.07274","repositories_listed":2,"syntology":null},{"url":"/paper/q-prop-sample-efficient-policy-gradient-with","slug":"q-prop-sample-efficient-policy-gradient-with","title":"Q-Prop: Sample-Efficient Policy Gradient with An Off-Policy Critic","date":"2016-11-07","arxiv_id":"1611.02247","repositories_listed":2,"syntology":null},{"url":"/paper/modular-multitask-reinforcement-learning-with","slug":"modular-multitask-reinforcement-learning-with","title":"Modular Multitask Reinforcement Learning with Policy Sketches","date":"2016-11-06","arxiv_id":"1611.01796","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/modular-multitask-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/1611.01796","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1611.01796"}},"official":{"repos":["jacobandreas/psketch"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/multi-objective-deep-reinforcement-learning","slug":"multi-objective-deep-reinforcement-learning","title":"Multi-Objective Deep Reinforcement Learning","date":"2016-10-09","arxiv_id":"1610.02707","repositories_listed":2,"syntology":null},{"url":"/paper/target-driven-visual-navigation-in-indoor","slug":"target-driven-visual-navigation-in-indoor","title":"Target-driven Visual Navigation in Indoor Scenes using Deep Reinforcement Learning","date":"2016-09-16","arxiv_id":"1609.05143","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-in-large-discrete","slug":"deep-reinforcement-learning-in-large-discrete","title":"Deep Reinforcement Learning in Large Discrete Action Spaces","date":"2015-12-24","arxiv_id":"1512.07679","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-gradient-1","slug":"deep-reinforcement-learning-with-gradient-1","title":"Deep Reinforcement Learning with Gradient Eligibility Traces","date":"2025-07-12","arxiv_id":"2507.09087","repositories_listed":1,"syntology":null},{"url":"/paper/generalized-adaptive-transfer-network","slug":"generalized-adaptive-transfer-network","title":"Generalized Adaptive Transfer Network: Enhancing Transfer Learning in Reinforcement Learning Across Domains","date":"2025-07-02","arxiv_id":"2507.03026","repositories_listed":1,"syntology":null},{"url":"/paper/traced-transition-aware-regret-approximation","slug":"traced-transition-aware-regret-approximation","title":"TRACED: Transition-aware Regret Approximation with Co-learnability for Environment Design","date":"2025-06-24","arxiv_id":"2506.19997","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/traced-transition-aware-regret-approximation#ran","syntology_url":"https://syntology.ai/paper/2506.19997","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.19997"}},"official":{"repos":["cho-geonwoo/traced"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-action-duration-with-contextual","slug":"adaptive-action-duration-with-contextual","title":"Adaptive Action Duration with Contextual Bandits for Deep Reinforcement Learning in Dynamic Environments","date":"2025-06-17","arxiv_id":"2507.00030","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-explore-in-diverse-reward","slug":"learning-to-explore-in-diverse-reward","title":"Learning to Explore in Diverse Reward Settings via Temporal-Difference-Error Maximization","date":"2025-06-16","arxiv_id":"2506.13345","repositories_listed":1,"syntology":null},{"url":"/paper/dr-sac-distributionally-robust-soft-actor","slug":"dr-sac-distributionally-robust-soft-actor","title":"DR-SAC: Distributionally Robust Soft Actor-Critic for Reinforcement Learning under Uncertainty","date":"2025-06-14","arxiv_id":"2506.12622","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/dr-sac-distributionally-robust-soft-actor#ran","syntology_url":"https://syntology.ai/paper/2506.12622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.12622"}},"official":{"repos":["lemutisme/dr-sac"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/2506-08344","slug":"2506-08344","title":"Re4MPC: Reactive Nonlinear MPC for Multi-model Motion Planning via Deep Reinforcement Learning","date":"2025-06-10","arxiv_id":"2506.08344","repositories_listed":1,"syntology":null},{"url":"/paper/from-static-to-adaptive-defense-federated","slug":"from-static-to-adaptive-defense-federated","title":"From Static to Adaptive Defense: Federated Multi-Agent Deep Reinforcement Learning-Driven Moving Target Defense Against DoS Attacks in UAV Swarm Networks","date":"2025-06-09","arxiv_id":"2506.07392","repositories_listed":1,"syntology":null},{"url":"/paper/realistic-urban-traffic-generator-using","slug":"realistic-urban-traffic-generator-using","title":"Realistic Urban Traffic Generator using Decentralized Federated Learning for the SUMO simulator","date":"2025-06-09","arxiv_id":"2506.07980","repositories_listed":1,"syntology":null},{"url":"/paper/cirl-open-source-environments-for","slug":"cirl-open-source-environments-for","title":"CiRL: Open-Source Environments for Reinforcement Learning in Circular Economy and Net Zero","date":"2025-05-24","arxiv_id":"2505.21536","repositories_listed":1,"syntology":null},{"url":"/paper/the-cell-must-go-on-agar-io-for-continual","slug":"the-cell-must-go-on-agar-io-for-continual","title":"The Cell Must Go On: Agar.io for Continual Reinforcement Learning","date":"2025-05-23","arxiv_id":"2505.18347","repositories_listed":1,"syntology":null},{"url":"/paper/graph-supported-dynamic-algorithm","slug":"graph-supported-dynamic-algorithm","title":"Graph-Supported Dynamic Algorithm Configuration for Multi-Objective Combinatorial Optimization","date":"2025-05-22","arxiv_id":"2505.16471","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/graph-supported-dynamic-algorithm#ran","syntology_url":"https://syntology.ai/paper/2505.16471","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16471"}},"official":{"repos":["robbertreijnen/gs-modac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gates-cost-aware-dynamic-workflow-scheduling","slug":"gates-cost-aware-dynamic-workflow-scheduling","title":"GATES: Cost-aware Dynamic Workflow Scheduling via Graph Attention Networks and Evolution Strategy","date":"2025-05-18","arxiv_id":"2505.12355","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/gates-cost-aware-dynamic-workflow-scheduling#ran","syntology_url":"https://syntology.ai/paper/2505.12355","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12355"}},"official":{"repos":["yashen998/gates"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/2505-11366","slug":"2505-11366","title":"Learning Multimodal AI Algorithms for Amplifying Limited User Input into High-dimensional Control Space","date":"2025-05-16","arxiv_id":"2505.11366","repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-on-a-budget-miniaturizing-deepseek","slug":"reasoning-on-a-budget-miniaturizing-deepseek","title":"Reasoning on a Budget: Miniaturizing DeepSeek R1 with SFT-GRPO Alignment for Instruction-Tuned LLMs","date":"2025-05-16","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-robustness-of-deep-reinforcement","slug":"evaluating-robustness-of-deep-reinforcement","title":"Evaluating Robustness of Deep Reinforcement Learning for Autonomous Surface Vehicle Control in Field Tests","date":"2025-05-15","arxiv_id":"2505.10033","repositories_listed":1,"syntology":null},{"url":"/paper/online-learning-based-adaptive-beam-switching","slug":"online-learning-based-adaptive-beam-switching","title":"Online Learning-based Adaptive Beam Switching for 6G Networks: Enhancing Efficiency and Resilience","date":"2025-05-12","arxiv_id":"2505.08032","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-cooperative-multi-agent","slug":"enhancing-cooperative-multi-agent","title":"Enhancing Cooperative Multi-Agent Reinforcement Learning with State Modelling and Adversarial Exploration","date":"2025-05-08","arxiv_id":"2505.05262","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":3,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":4,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/enhancing-cooperative-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2505.05262","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.05262"}},"official":{"repos":["ddaedalus/smpe"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/adaptive-and-robust-dbscan-with-multi-agent","slug":"adaptive-and-robust-dbscan-with-multi-agent","title":"Adaptive and Robust DBSCAN with Multi-agent Reinforcement Learning","date":"2025-05-07","arxiv_id":"2505.04339","repositories_listed":1,"syntology":null},{"url":"/paper/unraveling-the-rainbow-can-value-based","slug":"unraveling-the-rainbow-can-value-based","title":"Unraveling the Rainbow: can value-based methods schedule?","date":"2025-05-06","arxiv_id":"2505.03323","repositories_listed":1,"syntology":null},{"url":"/paper/graph-neural-network-based-reinforcement","slug":"graph-neural-network-based-reinforcement","title":"Graph Neural Network-Based Reinforcement Learning for Controlling Biological Networks: The GATTACA Framework","date":"2025-05-05","arxiv_id":"2505.02712","repositories_listed":1,"syntology":null},{"url":"/paper/automated-decision-making-for-dynamic-task","slug":"automated-decision-making-for-dynamic-task","title":"Automated decision-making for dynamic task assignment at scale","date":"2025-04-28","arxiv_id":"2504.19933","repositories_listed":1,"syntology":null},{"url":"/paper/neurophysiologically-realistic-environment","slug":"neurophysiologically-realistic-environment","title":"Neurophysiologically Realistic Environment for Comparing Adaptive Deep Brain Stimulation Algorithms in Parkinson Disease","date":"2025-04-26","arxiv_id":"2505.09624","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-sensor-steering-strategy-using-deep","slug":"adaptive-sensor-steering-strategy-using-deep","title":"Adaptive Sensor Steering Strategy Using Deep Reinforcement Learning for Dynamic Data Acquisition in Digital Twins","date":"2025-04-14","arxiv_id":"2504.10248","repositories_listed":1,"syntology":null},{"url":"/paper/pay-attention-to-what-and-where-interpretable","slug":"pay-attention-to-what-and-where-interpretable","title":"Pay Attention to What and Where? Interpretable Feature Extractor in Vision-based Deep Reinforcement Learning","date":"2025-04-14","arxiv_id":"2504.10071","repositories_listed":1,"syntology":null},{"url":"/paper/interq-a-dqn-framework-for-optimal","slug":"interq-a-dqn-framework-for-optimal","title":"InterQ: A DQN Framework for Optimal Intermittent Control","date":"2025-04-12","arxiv_id":"2504.09035","repositories_listed":1,"syntology":null},{"url":"/paper/trust-region-twisted-policy-improvement","slug":"trust-region-twisted-policy-improvement","title":"Trust-Region Twisted Policy Improvement","date":"2025-04-08","arxiv_id":"2504.06048","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforcement-learning-method-for","slug":"a-reinforcement-learning-method-for","title":"A Reinforcement Learning Method for Environments with Stochastic Variables: Post-Decision Proximal Policy Optimization with Dual Critic Networks","date":"2025-04-07","arxiv_id":"2504.05150","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-algorithms-for-1","slug":"deep-reinforcement-learning-algorithms-for-1","title":"Deep Reinforcement Learning Algorithms for Option Hedging","date":"2025-04-07","arxiv_id":"2504.05521","repositories_listed":1,"syntology":null},{"url":"/paper/ai2stow-end-to-end-deep-reinforcement","slug":"ai2stow-end-to-end-deep-reinforcement","title":"AI2STOW: End-to-End Deep Reinforcement Learning to Construct Master Stowage Plans under Demand Uncertainty","date":"2025-04-06","arxiv_id":"2504.04469","repositories_listed":1,"syntology":null},{"url":"/paper/generative-market-equilibrium-models-with","slug":"generative-market-equilibrium-models-with","title":"Generative Market Equilibrium Models with Stable Adversarial Learning via Reinforcement","date":"2025-04-05","arxiv_id":"2504.04300","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-via-object","slug":"deep-reinforcement-learning-via-object","title":"Deep Reinforcement Learning via Object-Centric Attention","date":"2025-04-03","arxiv_id":"2504.03024","repositories_listed":1,"syntology":null},{"url":"/paper/continual-reinforcement-learning-for-hvac","slug":"continual-reinforcement-learning-for-hvac","title":"Continual Reinforcement Learning for HVAC Systems Control: Integrating Hypernetworks and Transfer Learning","date":"2025-03-24","arxiv_id":"2503.19212","repositories_listed":1,"syntology":null},{"url":"/paper/whenever-wherever-towards-orchestrating-crowd","slug":"whenever-wherever-towards-orchestrating-crowd","title":"Whenever, Wherever: Towards Orchestrating Crowd Simulations with Spatio-Temporal Spawn Dynamics","date":"2025-03-20","arxiv_id":"2503.16639","repositories_listed":1,"syntology":null},{"url":"/paper/application-of-linear-regression-method-to","slug":"application-of-linear-regression-method-to","title":"Application of linear regression method to the deep reinforcement learning in continuous action cases","date":"2025-03-19","arxiv_id":"2503.14976","repositories_listed":1,"syntology":null},{"url":"/paper/ipcgrl-language-instructed-reinforcement","slug":"ipcgrl-language-instructed-reinforcement","title":"IPCGRL: Language-Instructed Reinforcement Learning for Procedural Level Generation","date":"2025-03-16","arxiv_id":"2503.12358","repositories_listed":1,"syntology":null},{"url":"/paper/mobility-aware-seamless-service-migration-and","slug":"mobility-aware-seamless-service-migration-and","title":"Mobility-aware Seamless Service Migration and Resource Allocation in Multi-edge IoV Systems","date":"2025-03-11","arxiv_id":"2503.13494","repositories_listed":1,"syntology":null},{"url":"/paper/learning-decision-trees-as-amortized","slug":"learning-decision-trees-as-amortized","title":"Learning Decision Trees as Amortized Structure Inference","date":"2025-03-10","arxiv_id":"2503.06985","repositories_listed":1,"syntology":null},{"url":"/paper/dynamics-invariant-quadrotor-control-using","slug":"dynamics-invariant-quadrotor-control-using","title":"Dynamics-Invariant Quadrotor Control using Scale-Aware Deep Reinforcement Learning","date":"2025-03-09","arxiv_id":"2503.09622","repositories_listed":1,"syntology":null},{"url":"/paper/studying-the-interplay-between-the-actor-and","slug":"studying-the-interplay-between-the-actor-and","title":"Studying the Interplay Between the Actor and Critic Representations in Reinforcement Learning","date":"2025-03-08","arxiv_id":"2503.06343","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/studying-the-interplay-between-the-actor-and#ran","syntology_url":"https://syntology.ai/paper/2503.06343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06343"}},"official":{"repos":["francelico/deac-rep"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/experience-replay-with-random-reshuffling","slug":"experience-replay-with-random-reshuffling","title":"Experience Replay with Random Reshuffling","date":"2025-03-04","arxiv_id":"2503.02269","repositories_listed":1,"syntology":null},{"url":"/paper/implementing-spiking-world-model-with-multi","slug":"implementing-spiking-world-model-with-multi","title":"Spiking World Model with Multi-Compartment Neurons for Model-based Reinforcement Learning","date":"2025-03-02","arxiv_id":"2503.00713","repositories_listed":1,"syntology":null},{"url":"/paper/colordynamic-generalizable-scalable-real-time","slug":"colordynamic-generalizable-scalable-real-time","title":"ColorDynamic: Generalizable, Scalable, Real-time, End-to-end Local Planner for Unstructured and Dynamic Environments","date":"2025-02-27","arxiv_id":"2502.19892","repositories_listed":1,"syntology":null},{"url":"/paper/highly-parallelized-reinforcement-learning","slug":"highly-parallelized-reinforcement-learning","title":"Highly Parallelized Reinforcement Learning Training with Relaxed Assignment Dependencies","date":"2025-02-27","arxiv_id":"2502.20190","repositories_listed":1,"syntology":null},{"url":"/paper/playing-pokemon-red-via-deep-reinforcement","slug":"playing-pokemon-red-via-deep-reinforcement","title":"Playing Pokémon Red via Deep Reinforcement Learning","date":"2025-02-27","arxiv_id":"2502.19920","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/playing-pokemon-red-via-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2502.19920","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.19920"}},"official":{"repos":["MarcoMeter/neroRL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/xss-adversarial-attacks-based-on-deep","slug":"xss-adversarial-attacks-based-on-deep","title":"XSS Adversarial Attacks Based on Deep Reinforcement Learning: A Replication and Extension Study","date":"2025-02-26","arxiv_id":"2502.19095","repositories_listed":1,"syntology":null},{"url":"/paper/controlling-dynamics-of-stochastic-systems","slug":"controlling-dynamics-of-stochastic-systems","title":"Controlling dynamics of stochastic systems with deep reinforcement learning","date":"2025-02-25","arxiv_id":"2502.18111","repositories_listed":1,"syntology":null},{"url":"/paper/fighter-jet-navigation-and-combat-using-deep","slug":"fighter-jet-navigation-and-combat-using-deep","title":"Fighter Jet Navigation and Combat using Deep Reinforcement Learning with Explainable AI","date":"2025-02-19","arxiv_id":"2502.13373","repositories_listed":1,"syntology":null},{"url":"/paper/learning-symbolic-task-decompositions-for","slug":"learning-symbolic-task-decompositions-for","title":"Learning Symbolic Task Decompositions for Multi-Agent Teams","date":"2025-02-19","arxiv_id":"2502.13376","repositories_listed":1,"syntology":null},{"url":"/paper/navigating-demand-uncertainty-in-container","slug":"navigating-demand-uncertainty-in-container","title":"Navigating Demand Uncertainty in Container Shipping: Deep Reinforcement Learning for Enabling Adaptive and Feasible Master Stowage Planning","date":"2025-02-18","arxiv_id":"2502.12756","repositories_listed":1,"syntology":null},{"url":"/paper/convex-is-back-solving-belief-mdps-with","slug":"convex-is-back-solving-belief-mdps-with","title":"Convex Is Back: Solving Belief MDPs With Convexity-Informed Deep Reinforcement Learning","date":"2025-02-13","arxiv_id":"2502.09298","repositories_listed":1,"syntology":null},{"url":"/paper/sigma-sheaf-informed-geometric-multi-agent","slug":"sigma-sheaf-informed-geometric-multi-agent","title":"SIGMA: Sheaf-Informed Geometric Multi-Agent Pathfinding","date":"2025-02-10","arxiv_id":"2502.06440","repositories_listed":1,"syntology":null},{"url":"/paper/policy-abstraction-and-nash-refinement-in","slug":"policy-abstraction-and-nash-refinement-in","title":"Policy Abstraction and Nash Refinement in Tree-Exploiting PSRO","date":"2025-02-05","arxiv_id":"2502.02901","repositories_listed":1,"syntology":null},{"url":"/paper/blood-glucose-level-prediction-in-type-1","slug":"blood-glucose-level-prediction-in-type-1","title":"Blood Glucose Level Prediction in Type 1 Diabetes Using Machine Learning","date":"2025-01-30","arxiv_id":"2502.00065","repositories_listed":1,"syntology":null},{"url":"/paper/neural-operator-based-reinforcement-learning","slug":"neural-operator-based-reinforcement-learning","title":"Neural Operator based Reinforcement Learning for Control of first-order PDEs with Spatially-Varying State Delay","date":"2025-01-30","arxiv_id":"2501.18201","repositories_listed":1,"syntology":null},{"url":"/paper/camp-in-the-odyssey-provably-robust","slug":"camp-in-the-odyssey-provably-robust","title":"CAMP in the Odyssey: Provably Robust Reinforcement Learning with Certified Radius Maximization","date":"2025-01-29","arxiv_id":"2501.17667","repositories_listed":1,"syntology":null},{"url":"/paper/a-comprehensive-survey-on-self-interpretable","slug":"a-comprehensive-survey-on-self-interpretable","title":"A Comprehensive Survey on Self-Interpretable Neural Networks","date":"2025-01-26","arxiv_id":"2501.15638","repositories_listed":1,"syntology":null},{"url":"/paper/unidoor-a-universal-framework-for-action","slug":"unidoor-a-universal-framework-for-action","title":"UNIDOOR: A Universal Framework for Action-Level Backdoor Attacks in Deep Reinforcement Learning","date":"2025-01-26","arxiv_id":"2501.15529","repositories_listed":1,"syntology":null},{"url":"/paper/divergence-augmented-policy-optimization-1","slug":"divergence-augmented-policy-optimization-1","title":"Divergence-Augmented Policy Optimization","date":"2025-01-25","arxiv_id":"2501.15034","repositories_listed":1,"syntology":null},{"url":"/paper/reducing-action-space-for-deep-reinforcement","slug":"reducing-action-space-for-deep-reinforcement","title":"Reducing Action Space for Deep Reinforcement Learning via Causal Effect Estimation","date":"2025-01-24","arxiv_id":"2501.14543","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-data-exploitation-in-deep","slug":"adaptive-data-exploitation-in-deep","title":"Adaptive Data Exploitation in Deep Reinforcement Learning","date":"2025-01-22","arxiv_id":"2501.12620","repositories_listed":1,"syntology":null},{"url":"/paper/forestprotector-an-iot-architecture","slug":"forestprotector-an-iot-architecture","title":"ForestProtector: An IoT Architecture Integrating Machine Vision and Deep Reinforcement Learning for Efficient Wildfire Monitoring","date":"2025-01-17","arxiv_id":"2501.09926","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-deep-reinforcement-learning-for-2","slug":"hierarchical-deep-reinforcement-learning-for-2","title":"Hierarchical Deep Reinforcement Learning for Adaptive Resource Management in Integrated Terrestrial and Non-Terrestrial Networks","date":"2025-01-16","arxiv_id":"2501.09212","repositories_listed":1,"syntology":null},{"url":"/paper/cheq-ing-the-box-safe-variable-impedance","slug":"cheq-ing-the-box-safe-variable-impedance","title":"CHEQ-ing the Box: Safe Variable Impedance Learning for Robotic Polishing","date":"2025-01-14","arxiv_id":"2501.07985","repositories_listed":1,"syntology":null},{"url":"/paper/cuasmrl-optimizing-gpu-sass-schedules-via","slug":"cuasmrl-optimizing-gpu-sass-schedules-via","title":"CuAsmRL: Optimizing GPU SASS Schedules via Deep Reinforcement Learning","date":"2025-01-14","arxiv_id":"2501.08071","repositories_listed":1,"syntology":null}],"record_sha256":"7d98fd7f4a04e6527341d86c05a6f7d6a8f59c2ab46fb5323da3c126de88bb17","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}