{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/sac/papers/2","list_of":"/method/sac","method":"SAC","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,168],"of":168,"counts":{"archive_papers_tagged":168,"with_a_code_link":69,"where_syntology_ran_a_sample":20,"not_listed_spam_title":0,"listed":168,"listed_where_code_ran":20,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":17,"every_run_a_failure_of_syntologys_instrument":3,"listed_with_a_run_with_no_instrument_failure":17,"listed_every_run_a_failure_of_syntologys_instrument":3,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/sac","prev":"/method/sac","next":null,"papers":[{"paper":"/paper/performance-comparison-of-deep-rl-algorithms","slug":"performance-comparison-of-deep-rl-algorithms","title":"Performance Comparison of Deep RL Algorithms for Energy Systems Optimal Scheduling","date":"2022-08-01","arxiv_id":"2208.00728","n_code_links":1,"syntology":null},{"paper":null,"slug":"value-function-decomposition-for-iterative","title":"Value Function Decomposition for Iterative Design of Reinforcement Learning Agents","date":"2022-06-24","arxiv_id":"2206.13901","n_code_links":0,"syntology":null},{"paper":null,"slug":"equivariant-reinforcement-learning-for","title":"Equivariant Reinforcement Learning for Quadrotor UAV","date":"2022-06-02","arxiv_id":"2206.01233","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-reward-poisoning-attacks-on-online","slug":"efficient-reward-poisoning-attacks-on-online","title":"Efficient Reward Poisoning Attacks on Online Deep Reinforcement Learning","date":"2022-05-30","arxiv_id":"2205.14842","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yinglunxu/reward_poisoning_attack_drl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"spatial-autoregressive-coding-for-graph","title":"Spatial Autoregressive Coding for Graph Neural Recommendation","date":"2022-05-19","arxiv_id":"2205.09489","n_code_links":0,"syntology":null},{"paper":null,"slug":"building-decision-forest-via-deep","title":"Building Decision Forest via Deep Reinforcement Learning","date":"2022-04-01","arxiv_id":"2204.00306","n_code_links":0,"syntology":null},{"paper":"/paper/microracer-a-didactic-environment-for-deep","slug":"microracer-a-didactic-environment-for-deep","title":"MicroRacer: a didactic environment for Deep Reinforcement Learning","date":"2022-03-20","arxiv_id":"2203.10494","n_code_links":1,"syntology":null},{"paper":null,"slug":"ai-based-robust-resource-allocation-in-end-to","title":"AI-based Robust Resource Allocation in End-to-End Network Slicing under Demand and CSI Uncertainties","date":"2022-02-10","arxiv_id":"2202.05131","n_code_links":0,"syntology":null},{"paper":"/paper/skrl-modular-and-flexible-library-for","slug":"skrl-modular-and-flexible-library-for","title":"skrl: Modular and Flexible Library for Reinforcement Learning","date":"2022-02-08","arxiv_id":"2202.03825","n_code_links":1,"syntology":null},{"paper":null,"slug":"super-reparametrizations-of-weighted-csps","title":"Super-Reparametrizations of Weighted CSPs: Properties and Optimization Perspective","date":"2022-01-06","arxiv_id":"2201.02018","n_code_links":0,"syntology":null},{"paper":"/paper/soft-actor-critic-with-cross-entropy-policy","slug":"soft-actor-critic-with-cross-entropy-policy","title":"Soft Actor-Critic with Cross-Entropy Policy Optimization","date":"2021-12-21","arxiv_id":"2112.11115","n_code_links":1,"syntology":null},{"paper":"/paper/stochastic-planner-actor-critic-for","slug":"stochastic-planner-actor-critic-for","title":"Stochastic Planner-Actor-Critic for Unsupervised Deformable Image Registration","date":"2021-12-14","arxiv_id":"2112.07415","n_code_links":1,"syntology":null},{"paper":"/paper/segment-and-complete-defending-object","slug":"segment-and-complete-defending-object","title":"Segment and Complete: Defending Object Detectors against Adversarial Patch Attacks with Robust Patch Detection","date":"2021-12-08","arxiv_id":"2112.04532","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["joellliu/segmentandcomplete"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"target-entropy-annealing-for-discrete-soft","title":"Target Entropy Annealing for Discrete Soft Actor-Critic","date":"2021-12-06","arxiv_id":"2112.02852","n_code_links":0,"syntology":null},{"paper":"/paper/mastering-atari-games-with-limited-data","slug":"mastering-atari-games-with-limited-data","title":"Mastering Atari Games with Limited Data","date":"2021-10-30","arxiv_id":"2111.00210","n_code_links":3,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["werner-duvaud/muzero-general","yewr/efficientzero"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/recurrent-off-policy-baselines-for-memory","slug":"recurrent-off-policy-baselines-for-memory","title":"Recurrent Off-policy Baselines for Memory-based Continuous Control","date":"2021-10-25","arxiv_id":"2110.12628","n_code_links":1,"syntology":null},{"paper":"/paper/balancing-value-underestimation-and","slug":"balancing-value-underestimation-and","title":"Balancing Value Underestimation and Overestimation with Realistic Actor-Critic","date":"2021-10-19","arxiv_id":"2110.09712","n_code_links":1,"syntology":null},{"paper":"/paper/continuous-control-with-action-quantization-1","slug":"continuous-control-with-action-quantization-1","title":"Continuous Control with Action Quantization from Demonstrations","date":"2021-10-19","arxiv_id":"2110.10149","n_code_links":1,"syntology":null},{"paper":"/paper/dropout-q-functions-for-doubly-efficient","slug":"dropout-q-functions-for-doubly-efficient","title":"Dropout Q-Functions for Doubly Efficient Reinforcement Learning","date":"2021-10-05","arxiv_id":"2110.02034","n_code_links":2,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["TakuyaHiraoka/Dropout-Q-Functions-for-Doubly-Efficient-Reinforcement-Learning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"paper":null,"slug":"parallel-actors-and-learners-a-framework-for","title":"Parallel Actors and Learners: A Framework for Generating Scalable RL Implementations","date":"2021-10-03","arxiv_id":"2110.01101","n_code_links":0,"syntology":null},{"paper":null,"slug":"causaldyna-improving-generalization-of-dyna","title":"CausalDyna: Improving Generalization of Dyna-style Reinforcement Learning via Counterfactual-Based Data Augmentation","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"experience-replay-more-when-it-s-a-key","title":"Experience Replay More When It's a Key Transition in Deep Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/explanation-aware-experience-replay-in-rule","slug":"explanation-aware-experience-replay-in-rule","title":"Explanation-Aware Experience Replay in Rule-Dense Environments","date":"2021-09-29","arxiv_id":"2109.14711","n_code_links":1,"syntology":null},{"paper":null,"slug":"faster-reinforcement-learning-with-value","title":"Faster Reinforcement Learning with Value Target Lower Bounding","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-controllable-elements-oriented","title":"Learning Controllable Elements Oriented Representations for Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"meta-attention-for-off-policy-actor-critic","title":"Meta Attention For Off-Policy Actor-Critic","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ovd-explorer-a-general-information-theoretic","title":"OVD-Explorer: A General Information-theoretic Exploration Approach for Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"spp-rl-state-planning-policy-reinforcement","title":"SPP-RL: State Planning Policy Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"improved-soft-actor-critic-mixing-prioritized","title":"Improved Soft Actor-Critic: Mixing Prioritized Off-Policy Samples with On-Policy Experience","date":"2021-09-24","arxiv_id":"2109.11767","n_code_links":0,"syntology":null},{"paper":null,"slug":"soft-actor-critic-with-integer-actions","title":"Soft Actor-Critic With Integer Actions","date":"2021-09-17","arxiv_id":"2109.08512","n_code_links":0,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-at-the-edge-of","slug":"deep-reinforcement-learning-at-the-edge-of","title":"Deep Reinforcement Learning at the Edge of the Statistical Precipice","date":"2021-08-30","arxiv_id":"2108.13264","n_code_links":3,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 4 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/rliable"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"wad-a-deep-reinforcement-learning-agent-for","title":"WAD: A Deep Reinforcement Learning Agent for Urban Autonomous Driving","date":"2021-08-27","arxiv_id":"2108.12134","n_code_links":0,"syntology":null},{"paper":null,"slug":"value-based-reinforcement-learning-for","title":"Value-Based Reinforcement Learning for Continuous Control Robotic Manipulation in Multi-Task Sparse Reward Settings","date":"2021-07-28","arxiv_id":"2107.13356","n_code_links":0,"syntology":null},{"paper":"/paper/brax-a-differentiable-physics-engine-for","slug":"brax-a-differentiable-physics-engine-for","title":"Brax -- A Differentiable Physics Engine for Large Scale Rigid Body Simulation","date":"2021-06-24","arxiv_id":"2106.13281","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google/brax"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-learning-based-optimal-market-bidding","title":"A Learning-based Optimal Market Bidding Strategy for Price-Maker Energy Storage","date":"2021-06-04","arxiv_id":"2106.02396","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-based-uav","title":"Deep Reinforcement Learning-based UAV Navigation and Control: A Soft Actor-Critic with Hindsight Experience Replay Approach","date":"2021-06-02","arxiv_id":"2106.01016","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-deeper-deep-reinforcement-learning","title":"Towards Deeper Deep Reinforcement Learning with Spectral Normalization","date":"2021-06-02","arxiv_id":"2106.01151","n_code_links":0,"syntology":null},{"paper":"/paper/context-based-soft-actor-critic-for","slug":"context-based-soft-actor-critic-for","title":"Context-Based Soft Actor Critic for Environments with Non-stationary Dynamics","date":"2021-05-07","arxiv_id":"2105.03310","n_code_links":1,"syntology":null},{"paper":null,"slug":"development-of-a-soft-actor-critic-deep","title":"Development of a Soft Actor Critic Deep Reinforcement Learning Approach for Harnessing Energy Flexibility in a Large Office Building","date":"2021-04-25","arxiv_id":"2104.12125","n_code_links":0,"syntology":null},{"paper":null,"slug":"acerac-efficient-reinforcement-learning-in","title":"ACERAC: Efficient reinforcement learning in fine time discretization","date":"2021-04-08","arxiv_id":"2104.04004","n_code_links":0,"syntology":null},{"paper":"/paper/identifying-co-adaptation-of-algorithmic-and","slug":"identifying-co-adaptation-of-algorithmic-and","title":"Co-Adaptation of Algorithmic and Implementational Innovations in Inference-based Deep Reinforcement Learning","date":"2021-03-31","arxiv_id":"2103.17258","n_code_links":1,"syntology":null},{"paper":null,"slug":"investigating-value-of-curriculum","title":"Investigating Value of Curriculum Reinforcement Learning in Autonomous Driving Under Diverse Road and Weather Conditions","date":"2021-03-14","arxiv_id":"2103.07903","n_code_links":0,"syntology":null},{"paper":null,"slug":"low-precision-reinforcement-learning","title":"Low-Precision Reinforcement Learning: Running Soft Actor-Critic in Half Precision","date":"2021-02-26","arxiv_id":"2102.13565","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-supervised-and-unsupervised-rewards","slug":"exploring-supervised-and-unsupervised-rewards","title":"Exploring Supervised and Unsupervised Rewards in Machine Translation","date":"2021-02-22","arxiv_id":"2102.11403","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-stage-transmission-line-flow-control","title":"Multi-Stage Transmission Line Flow Control Using Centralized and Decentralized Reinforcement Learning Agents","date":"2021-02-16","arxiv_id":"2102.08430","n_code_links":0,"syntology":null},{"paper":"/paper/q-value-weighted-regression-reinforcement-1","slug":"q-value-weighted-regression-reinforcement-1","title":"Q-Value Weighted Regression: Reinforcement Learning with Limited Data","date":"2021-02-12","arxiv_id":"2102.06782","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":null}},{"paper":"/paper/offcon-3-what-is-state-of-the-art-anyway","slug":"offcon-3-what-is-state-of-the-art-anyway","title":"OffCon$^3$: What is state of the art anyway?","date":"2021-01-27","arxiv_id":"2101.11331","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-semantic-adjacency-criterion-in-time","title":"The Semantic Adjacency Criterion in Time Intervals Mining","date":"2021-01-11","arxiv_id":"2101.03842","n_code_links":0,"syntology":null},{"paper":null,"slug":"cat-sac-soft-actor-critic-with-curiosity","title":"CAT-SAC: Soft Actor-Critic with Curiosity-Aware Entropy Temperature","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-coherent-exploration-for-continuous","title":"Deep Coherent Exploration For Continuous Control","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"pgps-coupling-policy-gradient-with-population","title":"PGPS : Coupling Policy Gradient with Population-based Search","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"2012-11643","title":"myGym: Modular Toolkit for Visuomotor Robotic Tasks","date":"2020-12-21","arxiv_id":"2012.11643","n_code_links":0,"syntology":null},{"paper":"/paper/policy-gradient-rl-algorithms-as-directed","slug":"policy-gradient-rl-algorithms-as-directed","title":"Policy Gradient RL Algorithms as Directed Acyclic Graphs","date":"2020-12-14","arxiv_id":"2012.07763","n_code_links":1,"syntology":null},{"paper":null,"slug":"virtual-autonomous-driving-with-reinforcement","title":"Virtual Autonomous Driving with Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"opac-opportunistic-actor-critic","title":"OPAC: Opportunistic Actor-Critic","date":"2020-12-11","arxiv_id":"2012.06555","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-reservoir-management-through-deep","title":"Efficient Reservoir Management through Deep Reinforcement Learning","date":"2020-12-07","arxiv_id":"2012.03822","n_code_links":0,"syntology":null},{"paper":"/paper/finrl-a-deep-reinforcement-learning-library","slug":"finrl-a-deep-reinforcement-learning-library","title":"FinRL: A Deep Reinforcement Learning Library for Automated Stock Trading in Quantitative Finance","date":"2020-11-19","arxiv_id":"2011.09607","n_code_links":6,"syntology":null},{"paper":"/paper/tonic-a-deep-reinforcement-learning-library","slug":"tonic-a-deep-reinforcement-learning-library","title":"Tonic: A Deep Reinforcement Learning Library for Fast Prototyping and Benchmarking","date":"2020-11-15","arxiv_id":"2011.07537","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["fabiopardo/tonic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/mesa-boost-ensemble-imbalanced-learning-with","slug":"mesa-boost-ensemble-imbalanced-learning-with","title":"MESA: Boost Ensemble Imbalanced Learning with MEta-SAmpler","date":"2020-10-17","arxiv_id":"2010.08830","n_code_links":2,"syntology":{"ran":15,"of":18,"n_ran_checked":11,"n_instrument":4,"unverified":3,"pointer_only":2,"phrase":"15 ran (of which 6 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 1 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","official":{"repos":["ZhiningLiu1998/mesa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/using-soft-actor-critic-for-low-level-uav","slug":"using-soft-actor-critic-for-low-level-uav","title":"Using Soft Actor-Critic for Low-Level UAV Control","date":"2020-10-05","arxiv_id":"2010.02293","n_code_links":1,"syntology":null},{"paper":"/paper/a-framework-for-reinforcement-learning-with","slug":"a-framework-for-reinforcement-learning-with","title":"A framework for reinforcement learning with autocorrelated actions","date":"2020-09-10","arxiv_id":"2009.04777","n_code_links":1,"syntology":null},{"paper":null,"slug":"measuring-the-credibility-of-student","title":"Measuring the Credibility of Student Attendance Data in Higher Education for Data Mining","date":"2020-09-01","arxiv_id":"2009.00679","n_code_links":0,"syntology":null},{"paper":"/paper/market-making-with-reinforcement-learning-sac","slug":"market-making-with-reinforcement-learning-sac","title":"Market-making with reinforcement-learning (SAC)","date":"2020-08-27","arxiv_id":"2008.12275","n_code_links":2,"syntology":null},{"paper":"/paper/evolve-to-control-evolution-based-soft-actor","slug":"evolve-to-control-evolution-based-soft-actor","title":"Maximum Mutation Reinforcement Learning for Scalable Control","date":"2020-07-24","arxiv_id":"2007.13690","n_code_links":2,"syntology":null},{"paper":"/paper/predictive-information-accelerates-learning","slug":"predictive-information-accelerates-learning","title":"Predictive Information Accelerates Learning in RL","date":"2020-07-24","arxiv_id":"2007.12401","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["google-research/pisac"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/meta-sac-auto-tune-the-entropy-temperature-of","slug":"meta-sac-auto-tune-the-entropy-temperature-of","title":"Meta-SAC: Auto-tune the Entropy Temperature of Soft Actor-Critic via Metagradient","date":"2020-07-03","arxiv_id":"2007.01932","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","official":{"repos":["twni2016/Meta-SAC"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/band-limited-soft-actor-critic-model","slug":"band-limited-soft-actor-critic-model","title":"Band-limited Soft Actor Critic Model","date":"2020-06-19","arxiv_id":"2006.11431","n_code_links":1,"syntology":null},{"paper":"/paper/detectors-detecting-objects-with-recursive-1","slug":"detectors-detecting-objects-with-recursive-1","title":"DetectoRS: Detecting Objects with Recursive Feature Pyramid and Switchable Atrous Convolution","date":"2020-06-03","arxiv_id":"2006.02334","n_code_links":6,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["joe-siyuan-qiao/DetectoRS"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}}],"record_sha256":"d18d733c50d52f3ad4620ad1f991278240f624ad7741e62cb5e08e88c36a7f4e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}