{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/75","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":75,"pages_in_order":152,"rows_per_page":100,"rows":[7401,7500],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/74","next":"/task/reinforcement-learning-1/papers/76","papers":[{"url":null,"slug":"autonomous-particles","title":"Autonomous particles","date":"2023-01-24","arxiv_id":"2301.10077","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-deep-reinforcement-learning-state","title":"Explainable Deep Reinforcement Learning: State of the Art and Challenges","date":"2023-01-24","arxiv_id":"2301.09937","repositories_listed":0,"syntology":null},{"url":null,"slug":"intrinsic-motivation-in-model-based","title":"Intrinsic Motivation in Model-based Reinforcement Learning: A Brief Review","date":"2023-01-24","arxiv_id":"2301.10067","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimal-value-equivalent-partial-models-for","title":"Minimal Value-Equivalent Partial Models for Scalable and Robust Planning in Lifelong Reinforcement Learning","date":"2023-01-24","arxiv_id":"2301.10119","repositories_listed":0,"syntology":null},{"url":null,"slug":"smart-self-supervised-multi-task-pretraining","title":"SMART: Self-supervised Multi-task pretrAining with contRol Transformers","date":"2023-01-24","arxiv_id":"2301.09816","repositories_listed":0,"syntology":null},{"url":null,"slug":"story-shaping-teaching-agents-human-like","title":"Story Shaping: Teaching Agents Human-like Behavior with Stories","date":"2023-01-24","arxiv_id":"2301.10107","repositories_listed":0,"syntology":null},{"url":null,"slug":"forecaster-aided-user-association-and-load","title":"Forecaster-aided User Association and Load Balancing in Multi-band Mobile Networks","date":"2023-01-23","arxiv_id":"2301.09294","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-view-decision-transformers-for","title":"Learning to View: Decision Transformers for Active Object Detection","date":"2023-01-23","arxiv_id":"2301.09544","repositories_listed":0,"syntology":null},{"url":null,"slug":"stock-trading-optimization-through-model-1","title":"Model Based Reinforcement Learning with Non-Gaussian Environment Dynamics and its Application to Portfolio Optimization","date":"2023-01-23","arxiv_id":"2301.09297","repositories_listed":0,"syntology":null},{"url":null,"slug":"quasi-optimal-learning-with-continuous","title":"Quasi-optimal Reinforcement Learning with Continuous Actions","date":"2023-01-21","arxiv_id":"2301.08940","repositories_listed":0,"syntology":null},{"url":null,"slug":"asynchronous-deep-double-duelling-q-learning","title":"Asynchronous Deep Double Duelling Q-Learning for Trading-Signal Execution in Limit Order Book Markets","date":"2023-01-20","arxiv_id":"2301.08688","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-slate-recommendation-with","title":"Generative Slate Recommendation with Reinforcement Learning","date":"2023-01-20","arxiv_id":"2301.08632","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-with-graph-2","title":"Multi-agent Reinforcement Learning with Graph Q-Networks for Antenna Tuning","date":"2023-01-20","arxiv_id":"2302.01199","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-armed-bandits-and-quantum-channel","title":"Multi-Armed Bandits and Quantum Channel Oracles","date":"2023-01-20","arxiv_id":"2301.08544","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-estimation-for","title":"Reinforcement learning-based estimation for partial differential equations","date":"2023-01-20","arxiv_id":"2302.01189","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-estimation-bias-in-policy","title":"Revisiting Estimation Bias in Policy Gradients for Deep Reinforcement Learning","date":"2023-01-20","arxiv_id":"2301.08442","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-meta-reinforcement-learning","title":"A Survey of Meta-Reinforcement Learning","date":"2023-01-19","arxiv_id":"2301.08028","repositories_listed":0,"syntology":null},{"url":null,"slug":"advanced-scaling-methods-for-vnf-deployment","title":"Advanced Scaling Methods for VNF deployment with Reinforcement Learning","date":"2023-01-19","arxiv_id":"2301.08325","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-gas-trading","title":"Domain-adapted Learning and Interpretability: DRL for Gas Trading","date":"2023-01-19","arxiv_id":"2301.08359","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-power-trading","title":"Domain-adapted Learning and Imitation: DRL for Power Arbitrage","date":"2023-01-19","arxiv_id":"2301.08360","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-diversity-in-unsupervised","title":"Generalization through Diversity: Improving Unsupervised Environment Design","date":"2023-01-19","arxiv_id":"2301.08025","repositories_listed":0,"syntology":null},{"url":null,"slug":"tight-guarantees-for-interactive-decision","title":"Tight Guarantees for Interactive Decision Making with the Decision-Estimation Coefficient","date":"2023-01-19","arxiv_id":"2301.08215","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-timescale-adaptation-in-an-open-ended","title":"Human-Timescale Adaptation in an Open-Ended Task Space","date":"2023-01-18","arxiv_id":"2301.07608","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-compartment-neuron-and-population","title":"Multi-compartment Neuron and Population Encoding Powered Spiking Neural Network for Deep Distributional Reinforcement Learning","date":"2023-01-18","arxiv_id":"2301.07275","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-robust-deep-reinforcement","title":"Adversarial Robust Deep Reinforcement Learning Requires Redefining Robustness","date":"2023-01-17","arxiv_id":"2301.07487","repositories_listed":0,"syntology":null},{"url":null,"slug":"dqnas-neural-architecture-search-using","title":"DQNAS: Neural Architecture Search using Reinforcement Learning","date":"2023-01-17","arxiv_id":"2301.06687","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-solve-arithmetic-problems-with-a","title":"Learning to solve arithmetic problems with a virtual abacus","date":"2023-01-17","arxiv_id":"2301.06870","repositories_listed":0,"syntology":null},{"url":null,"slug":"show-me-what-you-want-inverse-reinforcement","title":"Show me what you want: Inverse reinforcement learning to automatically design robot swarms by demonstration","date":"2023-01-17","arxiv_id":"2301.06864","repositories_listed":0,"syntology":null},{"url":null,"slug":"neuro-symbolic-world-models-for-adapting-to","title":"Neuro-Symbolic World Models for Adapting to Open World Novelty","date":"2023-01-16","arxiv_id":"2301.06294","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-human-cognition-with-a-hybrid-deep","title":"CogReact: A Reinforced Framework to Model Human Cognitive Reaction Modulated by Dynamic Intervention","date":"2023-01-15","arxiv_id":"2301.06216","repositories_listed":0,"syntology":null},{"url":null,"slug":"neuro-symbolic-meta-reinforcement-learning","title":"Neuro-symbolic Meta Reinforcement Learning for Trading","date":"2023-01-15","arxiv_id":"2302.08996","repositories_listed":0,"syntology":null},{"url":null,"slug":"first-three-years-of-the-international","title":"First Three Years of the International Verification of Neural Networks Competition (VNN-COMP)","date":"2023-01-14","arxiv_id":"2301.05815","repositories_listed":0,"syntology":null},{"url":null,"slug":"prudex-compass-towards-systematic-evaluation","title":"PRUDEX-Compass: Towards Systematic Evaluation of Reinforcement Learning in Financial Markets","date":"2023-01-14","arxiv_id":"2302.00586","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-protocol-synthesis","title":"Reinforcement Learning for Protocol Synthesis in Resource-Constrained Wireless Sensor and IoT Networks","date":"2023-01-14","arxiv_id":"2302.05300","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-averse-reinforcement-learning-via","title":"Risk-Averse Reinforcement Learning via Dynamic Time-Consistent Risk Measures","date":"2023-01-14","arxiv_id":"2301.05981","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-constrained-optimization-approach-to-the","title":"A Constrained-Optimization Approach to the Execution of Prioritized Stacks of Learned Multi-Robot Tasks","date":"2023-01-13","arxiv_id":"2301.05346","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-model-free-reinforcement","title":"Decentralized model-free reinforcement learning in stochastic games with average-reward objective","date":"2023-01-13","arxiv_id":"2301.05630","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-deep-q-learning-based-handover","title":"Hierarchical Deep Q-Learning Based Handover in Wireless Networks with Dual Connectivity","date":"2023-01-13","arxiv_id":"2301.05391","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-target-landmark-detection-with","title":"Multi-Target Landmark Detection with Incomplete Images via Reinforcement Learning and Shape Prior","date":"2023-01-13","arxiv_id":"2301.05392","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-sensitive-dead-end-identification-in","title":"Risk Sensitive Dead-end Identification in Safety-Critical Offline Reinforcement Learning","date":"2023-01-13","arxiv_id":"2301.05664","repositories_listed":0,"syntology":null},{"url":null,"slug":"asynchronous-training-of-quantum","title":"Asynchronous training of quantum reinforcement learning","date":"2023-01-12","arxiv_id":"2301.05096","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-joint-handover","title":"Reinforcement Learning-based Joint Handover and Beam Tracking in Millimeter-wave Networks","date":"2023-01-12","arxiv_id":"2301.05305","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-policy-improvement-for-pomdps-via-finite","title":"Safe Policy Improvement for POMDPs via Finite-State Controllers","date":"2023-01-12","arxiv_id":"2301.04939","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-analysis-of-quantile-temporal-difference","title":"An Analysis of Quantile Temporal-Difference Learning","date":"2023-01-11","arxiv_id":"2301.04462","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-preference-based-reinforcement","title":"Efficient Preference-Based Reinforcement Learning Using Learned Dynamics Models","date":"2023-01-11","arxiv_id":"2301.04741","repositories_listed":0,"syntology":null},{"url":null,"slug":"sok-adversarial-machine-learning-attacks-and","title":"SoK: Adversarial Machine Learning Attacks and Defences in Multi-Agent Reinforcement Learning","date":"2023-01-11","arxiv_id":"2301.04299","repositories_listed":0,"syntology":null},{"url":null,"slug":"switchable-lightweight-anti-symmetric","title":"Switchable Lightweight Anti-symmetric Processing (SLAP) with CNN Outspeeds Data Augmentation by Smaller Sample -- Application in Gomoku Reinforcement Learning","date":"2023-01-11","arxiv_id":"2301.04746","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-director-critic-a-novel-deep","title":"Actor-Director-Critic: A Novel Deep Reinforcement Learning Framework","date":"2023-01-10","arxiv_id":"2301.03887","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-ai-controlled-fes-restoration-of-arm","title":"Towards AI-controlled FES-restoration of arm movements: Controlling for progressive muscular fatigue with Gaussian state-space models","date":"2023-01-10","arxiv_id":"2301.04005","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-ai-controlled-fes-restoration-of-arm-1","title":"Towards AI-controlled FES-restoration of arm movements: neuromechanics-based reinforcement learning for 3-D reaching","date":"2023-01-10","arxiv_id":"2301.04004","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-in-model-based-reinforcement-1","title":"Exploration in Model-based Reinforcement Learning with Randomized Reward","date":"2023-01-09","arxiv_id":"2301.03142","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimax-weight-learning-for-absorbing-mdps","title":"Minimax Weight Learning for Absorbing MDPs","date":"2023-01-09","arxiv_id":"2301.03183","repositories_listed":0,"syntology":null},{"url":null,"slug":"network-slicing-via-transfer-learning-aided","title":"Network Slicing via Transfer Learning aided Distributed Deep Reinforcement Learning","date":"2023-01-09","arxiv_id":"2301.03262","repositories_listed":0,"syntology":null},{"url":null,"slug":"tuning-path-tracking-controllers-for","title":"Tuning Path Tracking Controllers for Autonomous Cars Using Reinforcement Learning","date":"2023-01-09","arxiv_id":"2301.03363","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-transformers-in-reinforcement","title":"A Survey on Transformers in Reinforcement Learning","date":"2023-01-08","arxiv_id":"2301.03044","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-symbolic-representations-for","title":"Learning Symbolic Representations for Reinforcement Learning of Non-Markovian Behavior","date":"2023-01-08","arxiv_id":"2301.02952","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-for-ris","title":"Hierarchical Reinforcement Learning for RIS-Assisted Energy-Efficient RAN","date":"2023-01-07","arxiv_id":"2301.02771","repositories_listed":0,"syntology":null},{"url":null,"slug":"laga-a-learning-adaptive-genetic-algorithm","title":"Mathematical Models and Reinforcement Learning based Evolutionary Algorithm Framework for Satellite Scheduling Problem","date":"2023-01-07","arxiv_id":"2301.02764","repositories_listed":0,"syntology":null},{"url":null,"slug":"markov-chain-concentration-with-an","title":"Markov Chain Concentration with an Application in Reinforcement Learning","date":"2023-01-07","arxiv_id":"2301.02926","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-based","title":"A Deep Reinforcement Learning-Based Controller for Magnetorheological-Damped Vehicle Suspension","date":"2023-01-06","arxiv_id":"2301.02714","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-for-fast","title":"Multi-Agent Reinforcement Learning for Fast-Timescale Demand Response of Residential Loads","date":"2023-01-06","arxiv_id":"2301.02593","repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-reset-free-reinforcement-learning-by","title":"Provable Reset-free Reinforcement Learning by No-Regret Reduction","date":"2023-01-06","arxiv_id":"2301.02389","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-inverse-reinforcement-learning","title":"Data-Driven Inverse Reinforcement Learning for Expert-Learner Zero-Sum Games","date":"2023-01-05","arxiv_id":"2301.01997","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-air-traffic","title":"Reinforcement Learning-Based Air Traffic Deconfliction","date":"2023-01-05","arxiv_id":"2301.01861","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-communication-for-multi-agent","title":"Scalable Communication for Multi-Agent Reinforcement Learning via Transformer-Based Email Mechanism","date":"2023-01-05","arxiv_id":"2301.01919","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-enhancement-of-reinforcement-learning","title":"Value Enhancement of Reinforcement Learning via Efficient and Robust Trust Region Optimization","date":"2023-01-05","arxiv_id":"2301.02220","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-based-mpc-from-big-data-using","title":"Learning-based MPC from Big Data Using Reinforcement Learning","date":"2023-01-04","arxiv_id":"2301.01667","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-aided-metaverse-over-wireless","title":"UAV aided Metaverse over Wireless Communications: A Reinforcement Learning Approach","date":"2023-01-04","arxiv_id":"2301.01474","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-succinct-summary-of-reinforcement-learning","title":"A Succinct Summary of Reinforcement Learning","date":"2023-01-03","arxiv_id":"2301.01379","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-conservative-q-learning-for","title":"Contextual Conservative Q-Learning for Offline Reinforcement Learning","date":"2023-01-03","arxiv_id":"2301.01298","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-evaluation-for-reinforcement-learning","title":"Offline Evaluation for Reinforcement Learning-based Recommendation: A Critical Issue and Some Alternatives","date":"2023-01-03","arxiv_id":"2301.00993","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-for-an-energy","title":"Safe Reinforcement Learning for an Energy-Efficient Driver Assistance System","date":"2023-01-03","arxiv_id":"2301.00904","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-difference-learning-with-compressed","title":"Temporal Difference Learning with Compressed Updates: Error-Feedback meets Reinforcement Learning","date":"2023-01-03","arxiv_id":"2301.00944","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-deployable-rl-what-s-broken-with-rl","title":"Towards Deployable RL - What's Broken with RL Research and a Potential Fix","date":"2023-01-03","arxiv_id":"2301.01320","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-rl-based-policy-optimization-method-guided","title":"A Policy Optimization Method Towards Optimal-time Stability","date":"2023-01-02","arxiv_id":"2301.00521","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-asset-1","title":"Deep Reinforcement Learning for Asset Allocation: Reward Clipping","date":"2023-01-02","arxiv_id":"2301.05300","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-traffic-signal-control-by-a-nash","title":"Large-Scale Traffic Signal Control by a Nash Deep Q-network Approach","date":"2023-01-02","arxiv_id":"2301.00637","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-filtering-for-reinforcement-learning","title":"Safety Filtering for Reinforcement Learning-based Adaptive Cruise Control","date":"2023-01-02","arxiv_id":"2301.00884","repositories_listed":0,"syntology":null},{"url":null,"slug":"co-speech-gesture-synthesis-by-reinforcement","title":"Co-Speech Gesture Synthesis by Reinforcement Learning With Contrastive Pre-Trained Rewards","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"local-guided-global-paired-similarity","title":"Local-Guided Global: Paired Similarity Representation for Visual Reinforcement Learning","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"optimization-of-image-transmission-in-a","title":"Optimization of Image Transmission in a Cooperative Semantic Communication Networks","date":"2023-01-01","arxiv_id":"2301.00433","repositories_listed":0,"syntology":null},{"url":null,"slug":"policycleanse-backdoor-detection-and","title":"PolicyCleanse: Backdoor Detection and Mitigation for Competitive Reinforcement Learning","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/second-thoughts-are-best-learning-to-re-align","slug":"second-thoughts-are-best-learning-to-re-align","title":"Second Thoughts are Best: Learning to Re-Align With Human Values from Text Edits","date":"2023-01-01","arxiv_id":"2301.00355","repositories_listed":0,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/second-thoughts-are-best-learning-to-re-align#ran","syntology_url":"https://syntology.ai/paper/2301.00355","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.00355"}},"official":null}},{"url":null,"slug":"simoun-synergizing-interactive-motion","title":"Simoun: Synergizing Interactive Motion-appearance Understanding for Vision-based Reinforcement Learning","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"stabilizing-visual-reinforcement-learning-via","title":"Stabilizing Visual Reinforcement Learning via Asymmetric Interactive Cooperation","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"accuracy-guaranteed-collaborative-dnn","title":"Accuracy-Guaranteed Collaborative DNN Inference in Industrial IoT via Deep Reinforcement Learning","date":"2022-12-31","arxiv_id":"2301.00130","repositories_listed":0,"syntology":null},{"url":null,"slug":"cost-effective-two-stage-network-slicing-for","title":"Cost-Effective Two-Stage Network Slicing for Edge-Cloud Orchestrated Vehicular Networks","date":"2022-12-31","arxiv_id":"2301.03358","repositories_listed":0,"syntology":null},{"url":null,"slug":"new-challenges-in-reinforcement-learning-a","title":"New Challenges in Reinforcement Learning: A Survey of Security and Privacy","date":"2022-12-31","arxiv_id":"2301.00188","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-deep-reinforcement-learning-and","title":"Hybrid Deep Reinforcement Learning and Planning for Safe and Comfortable Automated Driving","date":"2022-12-30","arxiv_id":"2301.00650","repositories_listed":0,"syntology":null},{"url":null,"slug":"pomrl-no-regret-learning-to-plan-with","title":"POMRL: No-Regret Learning-to-Plan with Increasing Horizons","date":"2022-12-30","arxiv_id":"2212.14530","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-experts-advice-aggregation-framework","title":"A Novel Experts Advice Aggregation Framework Using Deep Reinforcement Learning for Portfolio Management","date":"2022-12-29","arxiv_id":"2212.14477","repositories_listed":0,"syntology":null},{"url":null,"slug":"backward-curriculum-reinforcement-learning","title":"Backward Curriculum Reinforcement Learning","date":"2022-12-29","arxiv_id":"2212.14214","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-multi-agent-deep-reinforcement","title":"Federated Multi-Agent Deep Reinforcement Learning Approach via Physics-Informed Reward for Multi-Microgrid Energy Management","date":"2022-12-29","arxiv_id":"2301.00641","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-policy-optimization-in-rl-with","title":"Offline Policy Optimization in RL with Variance Regularizaton","date":"2022-12-29","arxiv_id":"2212.14405","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-geometry-of-reinforcement-learning-in","title":"On the Geometry of Reinforcement Learning in Continuous State and Action Spaces","date":"2022-12-29","arxiv_id":"2301.00009","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-transforming-reinforcement-learning-by","title":"On Transforming Reinforcement Learning by Transformer: The Development Trajectory","date":"2022-12-29","arxiv_id":"2212.14164","repositories_listed":0,"syntology":null},{"url":null,"slug":"certifying-safety-in-reinforcement-learning","title":"Certifying Safety in Reinforcement Learning under Adversarial Perturbation Attacks","date":"2022-12-28","arxiv_id":"2212.14115","repositories_listed":0,"syntology":null},{"url":null,"slug":"don-t-do-it-safer-reinforcement-learning-with","title":"Don't do it: Safer Reinforcement Learning With Rule-based Guidance","date":"2022-12-28","arxiv_id":"2212.13819","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-a-sequence-to-sequence-nlp-model","title":"Improving a sequence-to-sequence nlp model using a reinforcement learning policy algorithm","date":"2022-12-28","arxiv_id":"2212.14117","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-convergence-of-discounted-policy","title":"On the Convergence of Discounted Policy Gradient Methods","date":"2022-12-28","arxiv_id":"2212.14066","repositories_listed":0,"syntology":null}],"record_sha256":"dd2db394d6214d13b10c4ff933140e8443acd9852a8bf9ad2f260d1a4a320a03","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}