{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/39","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":39,"pages_in_order":152,"rows_per_page":100,"rows":[3801,3900],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/38","next":"/task/reinforcement-learning-1/papers/40","papers":[{"url":"/paper/a-distributional-view-on-multi-objective","slug":"a-distributional-view-on-multi-objective","title":"A Distributional View on Multi-Objective Policy Optimization","date":"2020-05-15","arxiv_id":"2005.07513","repositories_listed":1,"syntology":null},{"url":"/paper/think-too-fast-nor-too-slow-the-computational","slug":"think-too-fast-nor-too-slow-the-computational","title":"Think Too Fast Nor Too Slow: The Computational Trade-off Between Planning And Reinforcement Learning","date":"2020-05-15","arxiv_id":"2005.07404","repositories_listed":1,"syntology":null},{"url":"/paper/training-spiking-neural-networks-using","slug":"training-spiking-neural-networks-using","title":"Training spiking neural networks using reinforcement learning","date":"2020-05-12","arxiv_id":"2005.05941","repositories_listed":1,"syntology":null},{"url":"/paper/delay-aware-model-based-reinforcement","slug":"delay-aware-model-based-reinforcement","title":"Delay-Aware Model-Based Reinforcement Learning for Continuous Control","date":"2020-05-11","arxiv_id":"2005.05440","repositories_listed":1,"syntology":null},{"url":"/paper/delay-aware-multi-agent-reinforcement","slug":"delay-aware-multi-agent-reinforcement","title":"Delay-Aware Multi-Agent Reinforcement Learning for Cooperative and Competitive Environments","date":"2020-05-11","arxiv_id":"2005.05441","repositories_listed":1,"syntology":null},{"url":"/paper/mobile-robot-path-planning-in-dynamic","slug":"mobile-robot-path-planning-in-dynamic","title":"Mobile Robot Path Planning in Dynamic Environments through Globally Guided Reinforcement Learning","date":"2020-05-11","arxiv_id":"2005.05420","repositories_listed":1,"syntology":null},{"url":"/paper/unified-models-of-human-behavioral-agents-in","slug":"unified-models-of-human-behavioral-agents-in","title":"Unified Models of Human Behavioral Agents in Bandits, Contextual Bandits and RL","date":"2020-05-10","arxiv_id":"2005.04544","repositories_listed":1,"syntology":null},{"url":"/paper/allsteps-curriculum-driven-learning-of","slug":"allsteps-curriculum-driven-learning-of","title":"ALLSTEPS: Curriculum-driven Learning of Stepping Stone Skills","date":"2020-05-09","arxiv_id":"2005.04323","repositories_listed":1,"syntology":null},{"url":"/paper/learning-hierarchical-behavior-and-motion","slug":"learning-hierarchical-behavior-and-motion","title":"Learning hierarchical behavior and motion planning for autonomous driving","date":"2020-05-08","arxiv_id":"2005.03863","repositories_listed":1,"syntology":null},{"url":"/paper/carl-controllable-agent-with-reinforcement","slug":"carl-controllable-agent-with-reinforcement","title":"CARL: Controllable Agent with Reinforcement Learning for Quadruped Locomotion","date":"2020-05-07","arxiv_id":"2005.03288","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/carl-controllable-agent-with-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2005.03288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.03288"}},"official":{"repos":["inventec-ai-center/carl-siggraph2020"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/curious-hierarchical-actor-critic","slug":"curious-hierarchical-actor-critic","title":"Curious Hierarchical Actor-Critic Reinforcement Learning","date":"2020-05-07","arxiv_id":"2005.03420","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/curious-hierarchical-actor-critic#ran","syntology_url":"https://syntology.ai/paper/2005.03420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.03420"}},"official":{"repos":["knowledgetechnologyuhh/goal_conditioned_RL_baselines"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/plan2vec-unsupervised-representation-learning-1","slug":"plan2vec-unsupervised-representation-learning-1","title":"Plan2Vec: Unsupervised Representation Learning by Latent Plans","date":"2020-05-07","arxiv_id":"2005.03648","repositories_listed":1,"syntology":null},{"url":"/paper/supert-towards-new-frontiers-in-unsupervised","slug":"supert-towards-new-frontiers-in-unsupervised","title":"SUPERT: Towards New Frontiers in Unsupervised Evaluation Metrics for Multi-Document Summarization","date":"2020-05-07","arxiv_id":"2005.03724","repositories_listed":1,"syntology":null},{"url":"/paper/discrete-to-deep-supervised-policy-learning","slug":"discrete-to-deep-supervised-policy-learning","title":"Discrete-to-Deep Supervised Policy Learning","date":"2020-05-05","arxiv_id":"2005.02057","repositories_listed":1,"syntology":null},{"url":"/paper/gifting-in-multi-agent-reinforcement-learning","slug":"gifting-in-multi-agent-reinforcement-learning","title":"Gifting in multi-agent reinforcement learning","date":"2020-05-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-adversarial-inverse-reinforcement","slug":"off-policy-adversarial-inverse-reinforcement","title":"Off-Policy Adversarial Inverse Reinforcement Learning","date":"2020-05-03","arxiv_id":"2005.01138","repositories_listed":1,"syntology":null},{"url":"/paper/deep-symbolic-superoptimization-without-human","slug":"deep-symbolic-superoptimization-without-human","title":"Deep Symbolic Superoptimization Without Human Knowledge","date":"2020-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/explain-your-move-understanding-agent-actions","slug":"explain-your-move-understanding-agent-actions","title":"Explain Your Move: Understanding Agent Actions Using Focused Feature Saliency","date":"2020-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-collaborative-agents-with-rule","slug":"learning-collaborative-agents-with-rule","title":"Learning Collaborative Agents with Rule Guidance for Knowledge Graph Reasoning","date":"2020-05-01","arxiv_id":"2005.00571","repositories_listed":1,"syntology":null},{"url":"/paper/logic-and-the-2-simplicial-transformer-1","slug":"logic-and-the-2-simplicial-transformer-1","title":"Logic and the 2-Simplicial Transformer","date":"2020-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/option-discovery-using-deep-skill-chaining","slug":"option-discovery-using-deep-skill-chaining","title":"Option Discovery using Deep Skill Chaining","date":"2020-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/ract-toward-amortized-ranking-critical","slug":"ract-toward-amortized-ranking-critical","title":"RaCT: Toward Amortized Ranking-Critical Training For Collaborative Filtering","date":"2020-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/actor-critic-reinforcement-learning-for","slug":"actor-critic-reinforcement-learning-for","title":"Actor-Critic Reinforcement Learning for Control with Stability Guarantee","date":"2020-04-29","arxiv_id":"2004.14288","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/actor-critic-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2004.14288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.14288"}},"official":{"repos":["hithmh/Actor-critic-with-stability-guarantee"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/deep-reinforcement-learning-with-graph-based","slug":"deep-reinforcement-learning-with-graph-based","title":"Graph-based State Representation for Deep Reinforcement Learning","date":"2020-04-29","arxiv_id":"2004.13965","repositories_listed":1,"syntology":null},{"url":"/paper/transferable-active-grasping-and-real","slug":"transferable-active-grasping-and-real","title":"Transferable Active Grasping and Real Embodied Dataset","date":"2020-04-28","arxiv_id":"2004.13358","repositories_listed":1,"syntology":null},{"url":"/paper/evolving-inborn-knowledge-for-fast-adaptation","slug":"evolving-inborn-knowledge-for-fast-adaptation","title":"Evolving Inborn Knowledge For Fast Adaptation in Dynamic POMDP Problems","date":"2020-04-27","arxiv_id":"2004.12846","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-navigate-the-synthetically","slug":"learning-to-navigate-the-synthetically","title":"Learning To Navigate The Synthetically Accessible Chemical Space Using Reinforcement Learning","date":"2020-04-26","arxiv_id":"2004.12485","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-generalization-with","slug":"reinforcement-learning-generalization-with","title":"Reinforcement Learning Generalization with Surprise Minimization","date":"2020-04-26","arxiv_id":"2004.12399","repositories_listed":1,"syntology":null},{"url":"/paper/cfr-rl-traffic-engineering-with-reinforcement","slug":"cfr-rl-traffic-engineering-with-reinforcement","title":"CFR-RL: Traffic Engineering with Reinforcement Learning in SDN","date":"2020-04-24","arxiv_id":"2004.11986","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/cfr-rl-traffic-engineering-with-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2004.11986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.11986"}},"official":{"repos":["jrayzhang6/CFR-RL"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/curiosity-driven-energy-efficient-worker","slug":"curiosity-driven-energy-efficient-worker","title":"Curiosity-Driven Energy-Efficient Worker Scheduling in Vehicular Crowdsourcing: A Deep Reinforcement Learning Approach","date":"2020-04-24","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/self-paced-deep-reinforcement-learning","slug":"self-paced-deep-reinforcement-learning","title":"Self-Paced Deep Reinforcement Learning","date":"2020-04-24","arxiv_id":"2004.11812","repositories_listed":1,"syntology":null},{"url":"/paper/the-variational-bandwidth-bottleneck-1","slug":"the-variational-bandwidth-bottleneck-1","title":"The Variational Bandwidth Bottleneck: Stochastic Evaluation on an Information Budget","date":"2020-04-24","arxiv_id":"2004.11935","repositories_listed":1,"syntology":null},{"url":"/paper/correct-me-if-you-can-learning-from-error","slug":"correct-me-if-you-can-learning-from-error","title":"Correct Me If You Can: Learning from Error Corrections and Markings","date":"2020-04-23","arxiv_id":"2004.11222","repositories_listed":1,"syntology":null},{"url":"/paper/per-step-reward-a-new-perspective-for-risk","slug":"per-step-reward-a-new-perspective-for-risk","title":"Mean-Variance Policy Iteration for Risk-Averse Reinforcement Learning","date":"2020-04-22","arxiv_id":"2004.10888","repositories_listed":1,"syntology":null},{"url":"/paper/tactical-decision-making-in-autonomous","slug":"tactical-decision-making-in-autonomous","title":"Tactical Decision-Making in Autonomous Driving by Reinforcement Learning with Uncertainty Estimation","date":"2020-04-22","arxiv_id":"2004.10439","repositories_listed":1,"syntology":null},{"url":"/paper/energy-based-imitation-learning","slug":"energy-based-imitation-learning","title":"Energy-Based Imitation Learning","date":"2020-04-20","arxiv_id":"2004.09395","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/energy-based-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/2004.09395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.09395"}},"official":{"repos":["apexrl/EBIL-torch"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/self-guided-evolution-strategies-with","slug":"self-guided-evolution-strategies-with","title":"Self-Guided Evolution Strategies with Historical Estimated Gradients","date":"2020-04-20","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/analyzing-reinforcement-learning-benchmarks","slug":"analyzing-reinforcement-learning-benchmarks","title":"Analyzing Reinforcement Learning Benchmarks with Random Weight Guessing","date":"2020-04-16","arxiv_id":"2004.07707","repositories_listed":1,"syntology":null},{"url":"/paper/continual-reinforcement-learning-with-multi","slug":"continual-reinforcement-learning-with-multi","title":"Continual Reinforcement Learning with Multi-Timescale Replay","date":"2020-04-16","arxiv_id":"2004.07530","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/continual-reinforcement-learning-with-multi#ran","syntology_url":"https://syntology.ai/paper/2004.07530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.07530"}},"official":{"repos":["ChristosKap/multi_timescale_replay"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/fast-template-matching-and-update-for-video","slug":"fast-template-matching-and-update-for-video","title":"Fast Template Matching and Update for Video Object Tracking and Segmentation","date":"2020-04-16","arxiv_id":"2004.07538","repositories_listed":1,"syntology":null},{"url":"/paper/marleme-a-multi-agent-reinforcement-learning","slug":"marleme-a-multi-agent-reinforcement-learning","title":"MARLeME: A Multi-Agent Reinforcement Learning Model Extraction Library","date":"2020-04-16","arxiv_id":"2004.07928","repositories_listed":1,"syntology":null},{"url":"/paper/optigan-generative-adversarial-networks-for","slug":"optigan-generative-adversarial-networks-for","title":"OptiGAN: Generative Adversarial Networks for Goal Optimized Sequence Generation","date":"2020-04-16","arxiv_id":"2004.07534","repositories_listed":1,"syntology":null},{"url":"/paper/babyai-towards-grounded-language-learning","slug":"babyai-towards-grounded-language-learning","title":"Zero-Shot Compositional Policy Learning via Language Grounding","date":"2020-04-15","arxiv_id":"2004.07200","repositories_listed":1,"syntology":null},{"url":"/paper/prolog-technology-reinforcement-learning","slug":"prolog-technology-reinforcement-learning","title":"Prolog Technology Reinforcement Learning Prover","date":"2020-04-15","arxiv_id":"2004.06997","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-in-a-physics-inspired","slug":"reinforcement-learning-in-a-physics-inspired","title":"Reinforcement Learning in a Physics-Inspired Semi-Markov Environment","date":"2020-04-15","arxiv_id":"2004.07333","repositories_listed":1,"syntology":null},{"url":"/paper/a-text-based-deep-reinforcement-learning","slug":"a-text-based-deep-reinforcement-learning","title":"A Text-based Deep Reinforcement Learning Framework for Interactive Recommendation","date":"2020-04-14","arxiv_id":"2004.06651","repositories_listed":1,"syntology":null},{"url":"/paper/regret-bounds-for-kernel-based-reinforcement","slug":"regret-bounds-for-kernel-based-reinforcement","title":"Kernel-Based Reinforcement Learning: A Finite-Time Analysis","date":"2020-04-12","arxiv_id":"2004.05599","repositories_listed":1,"syntology":null},{"url":"/paper/self-punishment-and-reward-backfill-for-deep","slug":"self-punishment-and-reward-backfill-for-deep","title":"Self Punishment and Reward Backfill for Deep Q-Learning","date":"2020-04-10","arxiv_id":"2004.05002","repositories_listed":1,"syntology":null},{"url":"/paper/topological-quantum-compiling-with","slug":"topological-quantum-compiling-with","title":"Topological Quantum Compiling with Reinforcement Learning","date":"2020-04-09","arxiv_id":"2004.04743","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-transformers-in-rl","slug":"adaptive-transformers-in-rl","title":"Adaptive Transformers in RL","date":"2020-04-08","arxiv_id":"2004.03761","repositories_listed":1,"syntology":null},{"url":"/paper/continual-learning-with-gated-incremental-1","slug":"continual-learning-with-gated-incremental-1","title":"Continual Learning with Gated Incremental Memories for sequential data processing","date":"2020-04-08","arxiv_id":"2004.04077","repositories_listed":1,"syntology":null},{"url":"/paper/learning-from-learners-adapting-reinforcement","slug":"learning-from-learners-adapting-reinforcement","title":"Learning from Learners: Adapting Reinforcement Learning Agents to be Competitive in a Card Game","date":"2020-04-08","arxiv_id":"2004.04000","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-task-oriented-dialog-policy","slug":"multi-agent-task-oriented-dialog-policy","title":"Multi-Agent Task-Oriented Dialog Policy Learning with Role-Aware Reward Decomposition","date":"2020-04-08","arxiv_id":"2004.03809","repositories_listed":1,"syntology":null},{"url":"/paper/solving-the-scalarization-issues-of-advantage","slug":"solving-the-scalarization-issues-of-advantage","title":"Solving the scalarization issues of Advantage-based Reinforcement Learning Algorithms","date":"2020-04-08","arxiv_id":"2004.04120","repositories_listed":1,"syntology":null},{"url":"/paper/an-application-of-deep-reinforcement-learning","slug":"an-application-of-deep-reinforcement-learning","title":"An Application of Deep Reinforcement Learning to Algorithmic Trading","date":"2020-04-07","arxiv_id":"2004.06627","repositories_listed":1,"syntology":null},{"url":"/paper/guided-dialog-policy-learning-without","slug":"guided-dialog-policy-learning-without","title":"Guided Dialog Policy Learning without Adversarial Learning in the Loop","date":"2020-04-07","arxiv_id":"2004.03267","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/guided-dialog-policy-learning-without#ran","syntology_url":"https://syntology.ai/paper/2004.03267","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.03267"}},"official":{"repos":["cszmli/dp-without-adv"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/learning-2-opt-heuristics-for-the-traveling","slug":"learning-2-opt-heuristics-for-the-traveling","title":"Learning 2-opt Heuristics for the Traveling Salesman Problem via Deep Reinforcement Learning","date":"2020-04-03","arxiv_id":"2004.01608","repositories_listed":1,"syntology":null},{"url":"/paper/mri-reconstruction-with-interpretable-pixel","slug":"mri-reconstruction-with-interpretable-pixel","title":"MRI Reconstruction with Interpretable Pixel-Wise Operations Using Reinforcement Learning","date":"2020-04-03","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-for-2","slug":"multi-agent-reinforcement-learning-for-2","title":"Multi-agent Reinforcement Learning for Networked System Control","date":"2020-04-03","arxiv_id":"2004.01339","repositories_listed":1,"syntology":{"n":16,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":14,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":16,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 14 unverified","sample_list":"/paper/multi-agent-reinforcement-learning-for-2#ran","syntology_url":"https://syntology.ai/paper/2004.01339","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.01339"}},"official":{"repos":["cts198859/deeprl_network"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":14,"ran_from_kinds":["official"]}}},{"url":"/paper/action-space-shaping-in-deep-reinforcement","slug":"action-space-shaping-in-deep-reinforcement","title":"Action Space Shaping in Deep Reinforcement Learning","date":"2020-04-02","arxiv_id":"2004.00980","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/action-space-shaping-in-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2004.00980","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.00980"}},"official":{"repos":["Miffyli/rl-action-space-shaping"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/information-state-embedding-in-partially","slug":"information-state-embedding-in-partially","title":"Information State Embedding in Partially Observable Cooperative Multi-Agent Reinforcement Learning","date":"2020-04-02","arxiv_id":"2004.01098","repositories_listed":1,"syntology":null},{"url":"/paper/learning-sparse-rewarded-tasks-from-sub","slug":"learning-sparse-rewarded-tasks-from-sub","title":"Learning Sparse Rewarded Tasks from Sub-Optimal Demonstrations","date":"2020-04-01","arxiv_id":"2004.00530","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-sparse-rewarded-tasks-from-sub#ran","syntology_url":"https://syntology.ai/paper/2004.00530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.00530"}},"official":null}},{"url":"/paper/augmented-q-imitation-learning-aqil","slug":"augmented-q-imitation-learning-aqil","title":"Augmented Q Imitation Learning (AQIL)","date":"2020-03-31","arxiv_id":"2004.00993","repositories_listed":1,"syntology":null},{"url":"/paper/exploration-in-action-space","slug":"exploration-in-action-space","title":"Exploration in Action Space","date":"2020-03-31","arxiv_id":"2004.00500","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-ask-medical-questions-using","slug":"learning-to-ask-medical-questions-using","title":"Learning to Ask Medical Questions using Reinforcement Learning","date":"2020-03-31","arxiv_id":"2004.00994","repositories_listed":1,"syntology":null},{"url":"/paper/optimising-lockdown-policies-for-epidemic","slug":"optimising-lockdown-policies-for-epidemic","title":"Optimising Lockdown Policies for Epidemic Control using Reinforcement Learning","date":"2020-03-31","arxiv_id":"2003.14093","repositories_listed":1,"syntology":null},{"url":"/paper/straight-to-the-point-fast-forwarding-videos","slug":"straight-to-the-point-fast-forwarding-videos","title":"Straight to the Point: Fast-forwarding Videos via Reinforcement Learning Using Textual Data","date":"2020-03-31","arxiv_id":"2003.14229","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-large-scale","slug":"deep-reinforcement-learning-for-large-scale","title":"Deep reinforcement learning for large-scale epidemic control","date":"2020-03-30","arxiv_id":"2003.13676","repositories_listed":1,"syntology":null},{"url":"/paper/multi-task-reinforcement-learning-with-soft","slug":"multi-task-reinforcement-learning-with-soft","title":"Multi-Task Reinforcement Learning with Soft Modularization","date":"2020-03-30","arxiv_id":"2003.13661","repositories_listed":1,"syntology":null},{"url":"/paper/suphx-mastering-mahjong-with-deep","slug":"suphx-mastering-mahjong-with-deep","title":"Suphx: Mastering Mahjong with Deep Reinforcement Learning","date":"2020-03-30","arxiv_id":"2003.13590","repositories_listed":1,"syntology":null},{"url":"/paper/obstacle-avoidance-and-navigation-utilizing","slug":"obstacle-avoidance-and-navigation-utilizing","title":"Obstacle Avoidance and Navigation Utilizing Reinforcement Learning with Reward Shaping","date":"2020-03-28","arxiv_id":"2003.12863","repositories_listed":1,"syntology":null},{"url":"/paper/policy-teaching-via-environment-poisoning","slug":"policy-teaching-via-environment-poisoning","title":"Policy Teaching via Environment Poisoning: Training-time Adversarial Attacks against Reinforcement Learning","date":"2020-03-28","arxiv_id":"2003.12909","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/policy-teaching-via-environment-poisoning#ran","syntology_url":"https://syntology.ai/paper/2003.12909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.12909"}},"official":{"repos":["adishs/icml2020_rl-policy-teaching_code"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/machine-learning-in-asset-management-part-2","slug":"machine-learning-in-asset-management-part-2","title":"Machine Learning in Asset Management—Part 2: Portfolio Construction—Weight Optimization. The Journal of Financial Data Science","date":"2020-03-26","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/fiber-a-platform-for-efficient-development","slug":"fiber-a-platform-for-efficient-development","title":"Fiber: A Platform for Efficient Development and Distributed Training for Reinforcement Learning and Population-Based Methods","date":"2020-03-25","arxiv_id":"2003.11164","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fiber-a-platform-for-efficient-development#ran","syntology_url":"https://syntology.ai/paper/2003.11164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.11164"}},"official":null}},{"url":"/paper/an-empirical-investigation-of-the-challenges","slug":"an-empirical-investigation-of-the-challenges","title":"An empirical investigation of the challenges of real-world reinforcement learning","date":"2020-03-24","arxiv_id":"2003.11881","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/an-empirical-investigation-of-the-challenges#ran","syntology_url":"https://syntology.ai/paper/2003.11881","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.11881"}},"official":{"repos":["google-research/realworldrl_suite"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/evolutionary-population-curriculum-for-1","slug":"evolutionary-population-curriculum-for-1","title":"Evolutionary Population Curriculum for Scaling Multi-Agent Reinforcement Learning","date":"2020-03-23","arxiv_id":"2003.10423","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evolutionary-population-curriculum-for-1#ran","syntology_url":"https://syntology.ai/paper/2003.10423","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.10423"}},"official":{"repos":["qian18long/epciclr2020"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/using-deep-reinforcement-learning-methods-for","slug":"using-deep-reinforcement-learning-methods-for","title":"Using Deep Reinforcement Learning Methods for Autonomous Vessels in 2D Environments","date":"2020-03-23","arxiv_id":"2003.10249","repositories_listed":1,"syntology":null},{"url":"/paper/safe-reinforcement-learning-of-control-affine","slug":"safe-reinforcement-learning-of-control-affine","title":"Safe Reinforcement Learning of Control-Affine Systems with Vertex Networks","date":"2020-03-20","arxiv_id":"2003.09488","repositories_listed":1,"syntology":null},{"url":"/paper/adjust-planning-strategies-to-accommodate","slug":"adjust-planning-strategies-to-accommodate","title":"Adjust Planning Strategies to Accommodate Reinforcement Learning Agents","date":"2020-03-19","arxiv_id":"2003.08554","repositories_listed":1,"syntology":null},{"url":"/paper/enhanced-poet-open-ended-reinforcement","slug":"enhanced-poet-open-ended-reinforcement","title":"Enhanced POET: Open-Ended Reinforcement Learning through Unbounded Invention of Learning Challenges and their Solutions","date":"2020-03-19","arxiv_id":"2003.08536","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhanced-poet-open-ended-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2003.08536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.08536"}},"official":{"repos":["uber-research/poet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-fly-via-deep-model-based","slug":"learning-to-fly-via-deep-model-based","title":"Learning to Fly via Deep Model-Based Reinforcement Learning","date":"2020-03-19","arxiv_id":"2003.08876","repositories_listed":1,"syntology":null},{"url":"/paper/monotonic-value-function-factorisation-for","slug":"monotonic-value-function-factorisation-for","title":"Monotonic Value Function Factorisation for Deep Multi-Agent Reinforcement Learning","date":"2020-03-19","arxiv_id":"2003.08839","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/monotonic-value-function-factorisation-for#ran","syntology_url":"https://syntology.ai/paper/2003.08839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.08839"}},"official":{"repos":["oxwhirl/pymarl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/social-navigation-with-human-empowerment","slug":"social-navigation-with-human-empowerment","title":"Social Navigation with Human Empowerment driven Deep Reinforcement Learning","date":"2020-03-18","arxiv_id":"2003.08158","repositories_listed":1,"syntology":null},{"url":"/paper/giving-up-control-neurons-as-reinforcement","slug":"giving-up-control-neurons-as-reinforcement","title":"Giving Up Control: Neurons as Reinforcement Learning Agents","date":"2020-03-17","arxiv_id":"2003.11642","repositories_listed":1,"syntology":null},{"url":"/paper/simultaneous-navigation-and-radio-mapping-for","slug":"simultaneous-navigation-and-radio-mapping-for","title":"Simultaneous Navigation and Radio Mapping for Cellular-Connected UAV with Deep Reinforcement Learning","date":"2020-03-17","arxiv_id":"2003.07574","repositories_listed":1,"syntology":null},{"url":"/paper/particle-based-adaptive-discretization-for","slug":"particle-based-adaptive-discretization-for","title":"PFPN: Continuous Control of Physically Simulated Characters using Particle Filtering Policy Network","date":"2020-03-16","arxiv_id":"2003.06959","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-discovering-of-causal","slug":"self-supervised-discovering-of-causal","title":"Self-Supervised Discovering of Interpretable Features for Reinforcement Learning","date":"2020-03-16","arxiv_id":"2003.07069","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-supervised-discovering-of-causal#ran","syntology_url":"https://syntology.ai/paper/2003.07069","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.07069"}},"official":{"repos":["shiwj16/SSINet"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/provably-efficient-exploration-for-rl-with","slug":"provably-efficient-exploration-for-rl-with","title":"Provably Efficient Exploration for Reinforcement Learning Using Unsupervised Learning","date":"2020-03-15","arxiv_id":"2003.06898","repositories_listed":1,"syntology":null},{"url":"/paper/deep-deterministic-portfolio-optimization","slug":"deep-deterministic-portfolio-optimization","title":"Deep Deterministic Portfolio Optimization","date":"2020-03-13","arxiv_id":"2003.06497","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-deterministic-portfolio-optimization#ran","syntology_url":"https://syntology.ai/paper/2003.06497","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.06497"}},"official":{"repos":["CFMTech/Deep-RL-for-Portfolio-Optimization"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/option-discovery-in-the-absence-of-rewards","slug":"option-discovery-in-the-absence-of-rewards","title":"Option Discovery in the Absence of Rewards with Manifold Analysis","date":"2020-03-12","arxiv_id":"2003.05878","repositories_listed":1,"syntology":null},{"url":"/paper/the-chefs-hat-simulation-environment-for","slug":"the-chefs-hat-simulation-environment-for","title":"The Chef's Hat Simulation Environment for Reinforcement-Learning-Based Agents","date":"2020-03-12","arxiv_id":"2003.05861","repositories_listed":1,"syntology":null},{"url":"/paper/explore-and-exploit-with-heterotic-line","slug":"explore-and-exploit-with-heterotic-line","title":"Explore and Exploit with Heterotic Line Bundle Models","date":"2020-03-10","arxiv_id":"2003.04817","repositories_listed":1,"syntology":null},{"url":"/paper/stable-policy-optimization-via-off-policy","slug":"stable-policy-optimization-via-off-policy","title":"Stable Policy Optimization via Off-Policy Divergence Regularization","date":"2020-03-09","arxiv_id":"2003.04108","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stable-policy-optimization-via-off-policy#ran","syntology_url":"https://syntology.ai/paper/2003.04108","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.04108"}},"official":{"repos":["facebookresearch/ppo-dice"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-robustness-of-cooperative-multi-agent","slug":"on-the-robustness-of-cooperative-multi-agent","title":"On the Robustness of Cooperative Multi-Agent Reinforcement Learning","date":"2020-03-08","arxiv_id":"2003.03722","repositories_listed":1,"syntology":null},{"url":"/paper/ig-rl-inductive-graph-reinforcement-learning","slug":"ig-rl-inductive-graph-reinforcement-learning","title":"IG-RL: Inductive Graph Reinforcement Learning for Massive-Scale Traffic Signal Control","date":"2020-03-06","arxiv_id":"2003.05738","repositories_listed":1,"syntology":null},{"url":"/paper/can-increasing-input-dimensionality-improve","slug":"can-increasing-input-dimensionality-improve","title":"Can Increasing Input Dimensionality Improve Deep Reinforcement Learning?","date":"2020-03-03","arxiv_id":"2003.01629","repositories_listed":1,"syntology":null},{"url":"/paper/contention-window-optimization-in-ieee","slug":"contention-window-optimization-in-ieee","title":"Contention Window Optimization in IEEE 802.11ax Networks with Deep Reinforcement Learning","date":"2020-03-03","arxiv_id":"2003.01492","repositories_listed":1,"syntology":null},{"url":"/paper/embodied-synaptic-plasticity-with-online","slug":"embodied-synaptic-plasticity-with-online","title":"Embodied Synaptic Plasticity with Online Reinforcement learning","date":"2020-03-03","arxiv_id":"2003.01431","repositories_listed":1,"syntology":null},{"url":"/paper/robust-market-making-via-adversarial","slug":"robust-market-making-via-adversarial","title":"Robust Market Making via Adversarial Reinforcement Learning","date":"2020-03-03","arxiv_id":"2003.01820","repositories_listed":1,"syntology":null},{"url":"/paper/autophase-juggling-hls-phase-orderings-in","slug":"autophase-juggling-hls-phase-orderings-in","title":"AutoPhase: Juggling HLS Phase Orderings in Random Forests with Deep Reinforcement Learning","date":"2020-03-02","arxiv_id":"2003.00671","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autophase-juggling-hls-phase-orderings-in#ran","syntology_url":"https://syntology.ai/paper/2003.00671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.00671"}},"official":{"repos":["ucb-bar/autophase"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"167da97d3de526ed37ca5c54bc07ef691fa53a2c816ca85bac715f7a6be55e01","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}