{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/monte-carlo-tree-search/papers/2","list_of":"/method/monte-carlo-tree-search","method":"Monte-Carlo Tree Search","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,166],"of":166,"counts":{"archive_papers_tagged":166,"with_a_code_link":62,"where_syntology_ran_a_sample":12,"not_listed_spam_title":0,"listed":166,"listed_where_code_ran":12,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":10,"every_run_a_failure_of_syntologys_instrument":2,"listed_with_a_run_with_no_instrument_failure":10,"listed_every_run_a_failure_of_syntologys_instrument":2,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/monte-carlo-tree-search","prev":"/method/monte-carlo-tree-search","next":null,"papers":[{"paper":null,"slug":"self-play-learning-strategies-for-resource","title":"Self-play Learning Strategies for Resource Assignment in Open-RAN Networks","date":"2021-03-03","arxiv_id":"2103.02649","n_code_links":0,"syntology":null},{"paper":"/paper/visualizing-muzero-models","slug":"visualizing-muzero-models","title":"Visualizing MuZero Models","date":"2021-02-25","arxiv_id":"2102.12924","n_code_links":1,"syntology":null},{"paper":null,"slug":"combining-off-and-on-policy-training-in-model","title":"Combining Off and On-Policy Training in Model-Based Reinforcement Learning","date":"2021-02-24","arxiv_id":"2102.12194","n_code_links":0,"syntology":null},{"paper":"/paper/improving-model-based-reinforcement-learning","slug":"improving-model-based-reinforcement-learning","title":"Improving Model-Based Reinforcement Learning with Internal State Representations through Self-Supervision","date":"2021-02-10","arxiv_id":"2102.05599","n_code_links":2,"syntology":null},{"paper":"/paper/deep-learning-for-general-game-playing-with","slug":"deep-learning-for-general-game-playing-with","title":"Deep Learning for General Game Playing with Ludii and Polygames","date":"2021-01-23","arxiv_id":"2101.09562","n_code_links":1,"syntology":null},{"paper":null,"slug":"monte-carlo-planning-and-learning-with","title":"Monte-Carlo Planning and Learning with Language Action Value Estimates","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"playing-nondeterministic-games-through","title":"Playing Nondeterministic Games through Planning with a Learned Model","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/monte-carlo-graph-search-for-alphazero","slug":"monte-carlo-graph-search-for-alphazero","title":"Monte-Carlo Graph Search for AlphaZero","date":"2020-12-20","arxiv_id":"2012.11045","n_code_links":3,"syntology":null},{"paper":null,"slug":"driving-policy-adaptive-safeguard-for","title":"Driving-Policy Adaptive Safeguard for Autonomous Vehicles Using Reinforcement Learning","date":"2020-12-02","arxiv_id":"2012.01010","n_code_links":0,"syntology":null},{"paper":"/paper/hierarchical-clustering-in-particle-physics","slug":"hierarchical-clustering-in-particle-physics","title":"Hierarchical clustering in particle physics through reinforcement learning","date":"2020-11-16","arxiv_id":"2011.08191","n_code_links":1,"syntology":null},{"paper":null,"slug":"critic-pi2-master-continuous-planning-via","title":"Critic PI2: Master Continuous Planning via Policy Improvement with Path Integrals and Deep Actor-Critic Reinforcement Learning","date":"2020-11-13","arxiv_id":"2011.06752","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-role-of-planning-in-model-based-deep-1","title":"On the role of planning in model-based deep reinforcement learning","date":"2020-11-08","arxiv_id":"2011.04021","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-value-equivalence-principle-for-model","title":"The Value Equivalence Principle for Model-Based Reinforcement Learning","date":"2020-11-06","arxiv_id":"2011.03506","n_code_links":0,"syntology":null},{"paper":"/paper/interleaving-fast-and-slow-decision-making","slug":"interleaving-fast-and-slow-decision-making","title":"Interleaving Fast and Slow Decision Making","date":"2020-10-30","arxiv_id":"2010.16244","n_code_links":1,"syntology":null},{"paper":"/paper/dream-and-search-to-control-latent-space-1","slug":"dream-and-search-to-control-latent-space-1","title":"Dream and Search to Control: Latent Space Planning for Continuous Control","date":"2020-10-19","arxiv_id":"2010.09832","n_code_links":1,"syntology":null},{"paper":null,"slug":"alphazero-based-post-storm-vehicle-routing","title":"AlphaZero Based Post-Storm Repair Crew Dispatch for Distribution Grid Restoration","date":"2020-10-14","arxiv_id":"2010.06764","n_code_links":0,"syntology":null},{"paper":null,"slug":"playing-carcassonne-with-monte-carlo-tree","title":"Playing Carcassonne with Monte Carlo Tree Search","date":"2020-09-27","arxiv_id":"2009.12974","n_code_links":0,"syntology":null},{"paper":null,"slug":"formal-fields-a-framework-to-automate-code","title":"Formal Fields: A Framework to Automate Code Generation Across Domains","date":"2020-07-28","arxiv_id":"2007.14075","n_code_links":0,"syntology":null},{"paper":"/paper/monte-carlo-tree-search-as-regularized-policy","slug":"monte-carlo-tree-search-as-regularized-policy","title":"Monte-Carlo Tree Search as Regularized Policy Optimization","date":"2020-07-24","arxiv_id":"2007.12509","n_code_links":3,"syntology":null},{"paper":"/paper/the-loca-regret-a-consistent-metric-to","slug":"the-loca-regret-a-consistent-metric-to","title":"The LoCA Regret: A Consistent Metric to Evaluate Model-Based Behavior in Reinforcement Learning","date":"2020-07-07","arxiv_id":"2007.03158","n_code_links":2,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["chandar-lab/LoCA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"introduction-to-behavior-algorithms-for","title":"Introduction to Behavior Algorithms for Fighting Games","date":"2020-07-06","arxiv_id":"2007.12586","n_code_links":0,"syntology":null},{"paper":null,"slug":"convex-regularization-in-monte-carlo-tree","title":"Convex Regularization in Monte-Carlo Tree Search","date":"2020-07-01","arxiv_id":"2007.00391","n_code_links":0,"syntology":null},{"paper":null,"slug":"practical-large-scale-distributed-parallel","title":"Practical Massively Parallel Monte-Carlo Tree Search Applied to Molecular Design","date":"2020-06-18","arxiv_id":"2006.10504","n_code_links":0,"syntology":null},{"paper":null,"slug":"continuous-control-for-searching-and-planning","title":"Continuous Control for Searching and Planning with a Learned Model","date":"2020-06-12","arxiv_id":"2006.07430","n_code_links":0,"syntology":null},{"paper":null,"slug":"starcraft-ii-build-order-optimization-using","title":"StarCraft II Build Order Optimization using Deep Reinforcement Learning and Monte-Carlo Tree Search","date":"2020-06-12","arxiv_id":"2006.10525","n_code_links":0,"syntology":null},{"paper":null,"slug":"planning-in-markov-decision-processes-with","title":"Planning in Markov Decision Processes with Gap-Dependent Sample Complexity","date":"2020-06-10","arxiv_id":"2006.05879","n_code_links":0,"syntology":null},{"paper":null,"slug":"poly-hoot-monte-carlo-planning-in-continuous","title":"POLY-HOOT: Monte-Carlo Planning in Continuous Space MDPs with Non-Asymptotic Analysis","date":"2020-06-08","arxiv_id":"2006.04672","n_code_links":0,"syntology":null},{"paper":"/paper/manipulating-the-distributions-of-experience","slug":"manipulating-the-distributions-of-experience","title":"Manipulating the Distributions of Experience used for Self-Play Learning in Expert Iteration","date":"2020-05-30","arxiv_id":"2006.00283","n_code_links":1,"syntology":null},{"paper":null,"slug":"unlucky-explorer-a-complete-non-overlapping","title":"Unlucky Explorer: A Complete non-Overlapping Map Exploration","date":"2020-05-28","arxiv_id":"2005.14156","n_code_links":0,"syntology":null},{"paper":null,"slug":"single-agent-optimization-through-policy","title":"Single-Agent Optimization Through Policy Iteration Using Monte-Carlo Tree Search","date":"2020-05-22","arxiv_id":"2005.11335","n_code_links":0,"syntology":null},{"paper":"/paper/neural-machine-translation-with-monte-carlo","slug":"neural-machine-translation-with-monte-carlo","title":"Neural Machine Translation with Monte-Carlo Tree Search","date":"2020-04-27","arxiv_id":"2004.12527","n_code_links":1,"syntology":null},{"paper":"/paper/prolog-technology-reinforcement-learning","slug":"prolog-technology-reinforcement-learning","title":"Prolog Technology Reinforcement Learning Prover","date":"2020-04-15","arxiv_id":"2004.06997","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhanced-rolling-horizon-evolution-algorithm","title":"Enhanced Rolling Horizon Evolution Algorithm with Opponent Model Learning: Results for the Fighting Game AI Competition","date":"2020-03-31","arxiv_id":"2003.13949","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-reinforcement-learning-for-turn-based-zero","title":"On Reinforcement Learning for Turn-based Zero-sum Markov Games","date":"2020-02-25","arxiv_id":"2002.10620","n_code_links":0,"syntology":null},{"paper":null,"slug":"service-selection-using-predictive-models-and","title":"Service Selection using Predictive Models and Monte-Carlo Tree Search","date":"2020-02-12","arxiv_id":"2002.04852","n_code_links":0,"syntology":null},{"paper":null,"slug":"bayesian-optimization-for-backpropagation-in","title":"Bayesian optimization for backpropagation in Monte-Carlo tree search","date":"2020-01-25","arxiv_id":"2001.09325","n_code_links":0,"syntology":null},{"paper":null,"slug":"monte-carlo-tree-search-for-policy","title":"Monte-Carlo Tree Search for Policy Optimization","date":"2019-12-23","arxiv_id":"1912.10648","n_code_links":0,"syntology":null},{"paper":null,"slug":"combining-q-learning-and-search-with-1","title":"Combining Q-Learning and Search with Amortized Value Estimates","date":"2019-12-05","arxiv_id":"1912.02807","n_code_links":0,"syntology":null},{"paper":null,"slug":"maximum-entropy-monte-carlo-planning","title":"Maximum Entropy Monte-Carlo Planning","date":"2019-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/mastering-atari-go-chess-and-shogi-by","slug":"mastering-atari-go-chess-and-shogi-by","title":"Mastering Atari, Go, Chess and Shogi by Planning with a Learned Model","date":"2019-11-19","arxiv_id":"1911.08265","n_code_links":18,"syntology":{"ran":43,"of":64,"n_ran_checked":43,"n_instrument":0,"unverified":21,"pointer_only":62,"phrase":"43 ran (of which 36 constructed an object rather than computing a result; 43 with no instrument failure: 4 honoured, 0 violated, 39 with no contract checked; 0 where Syntology's instrument failed) · 21 unverified","official":null}},{"paper":null,"slug":"generalized-mean-estimation-in-monte-carlo","title":"Generalized Mean Estimation in Monte-Carlo Tree Search","date":"2019-11-01","arxiv_id":"1911.00384","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-multivariate-bandit-algorithm-with","title":"Efficient Multivariate Bandit Algorithm with Path Planning","date":"2019-09-06","arxiv_id":"1909.02705","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-play-the-chess-variant-crazyhouse","slug":"learning-to-play-the-chess-variant-crazyhouse","title":"Learning to play the Chess Variant Crazyhouse above World Champion Level with Deep Neural Networks and Human Data","date":"2019-08-19","arxiv_id":"1908.06660","n_code_links":3,"syntology":null},{"paper":"/paper/190600170","slug":"190600170","title":"Automated Machine Learning with Monte-Carlo Tree Search","date":"2019-06-01","arxiv_id":"1906.00170","n_code_links":2,"syntology":null},{"paper":null,"slug":"on-value-functions-and-the-agent-environment","title":"On Value Functions and the Agent-Environment Boundary","date":"2019-05-30","arxiv_id":"1905.13341","n_code_links":0,"syntology":null},{"paper":"/paper/misleading-authorship-attribution-of-source","slug":"misleading-authorship-attribution-of-source","title":"Misleading Authorship Attribution of Source Code using Adversarial Learning","date":"2019-05-29","arxiv_id":"1905.12386","n_code_links":1,"syntology":null},{"paper":null,"slug":"understanding-generalizing-alphago-zero","title":"Understanding & Generalizing AlphaGo Zero","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/monte-carlo-tree-search-for-efficient","slug":"monte-carlo-tree-search-for-efficient","title":"Monte-Carlo Tree Search for Efficient Visually Guided Rearrangement Planning","date":"2019-04-23","arxiv_id":"1904.10348","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ylabbe/rearrangement-planning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"structured-agents-for-physical-construction","title":"Structured agents for physical construction","date":"2019-04-05","arxiv_id":"1904.03177","n_code_links":0,"syntology":null},{"paper":null,"slug":"bayesian-reinforcement-learning-in-factored","title":"Bayesian Reinforcement Learning in Factored POMDPs","date":"2018-11-14","arxiv_id":"1811.05612","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-hearthstone-ai-by-combining-mcts","title":"Improving Hearthstone AI by Combining MCTS and Supervised Learning Algorithms","date":"2018-08-14","arxiv_id":"1808.04794","n_code_links":0,"syntology":null},{"paper":"/paper/hierarchical-reinforcement-learning-for-zero","slug":"hierarchical-reinforcement-learning-for-zero","title":"Hierarchical Reinforcement Learning for Zero-shot Generalization with Subtask Dependencies","date":"2018-07-19","arxiv_id":"1807.07665","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["srsohn/subtask-graph-execution"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"feedback-based-tree-search-for-reinforcement","title":"Feedback-Based Tree Search for Reinforcement Learning","date":"2018-05-15","arxiv_id":"1805.05935","n_code_links":0,"syntology":null},{"paper":null,"slug":"active-reinforcement-learning-with-monte","title":"Active Reinforcement Learning with Monte-Carlo Tree Search","date":"2018-03-13","arxiv_id":"1803.04926","n_code_links":0,"syntology":null},{"paper":null,"slug":"q-cp-learning-action-values-for-cooperative","title":"Q-CP: Learning Action Values for Cooperative Planning","date":"2018-03-01","arxiv_id":"1803.00297","n_code_links":0,"syntology":null},{"paper":null,"slug":"latent-forward-model-for-real-time-strategy","title":"Latent forward model for Real-time Strategy game planning with incomplete information","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-task-graph-execution","title":"Neural Task Graph Execution","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-motion-gaming-ai-for-health","title":"Adaptive Motion Gaming AI for Health Promotion","date":"2017-04-04","arxiv_id":"1704.00961","n_code_links":0,"syntology":null},{"paper":"/paper/generalised-discount-functions-applied-to-a","slug":"generalised-discount-functions-applied-to-a","title":"Generalised Discount Functions applied to a Monte-Carlo AImu Implementation","date":"2017-03-03","arxiv_id":"1703.01358","n_code_links":1,"syntology":null},{"paper":null,"slug":"monte-carlo-planning-method-estimates","title":"Monte Carlo Planning method estimates planning horizons during interactive social exchange","date":"2015-02-12","arxiv_id":"1502.03696","n_code_links":0,"syntology":null},{"paper":"/paper/move-evaluation-in-go-using-deep","slug":"move-evaluation-in-go-using-deep","title":"Move Evaluation in Go Using Deep Convolutional Neural Networks","date":"2014-12-20","arxiv_id":"1412.6564","n_code_links":1,"syntology":null},{"paper":null,"slug":"monte-carlo-planning-theoretically-fast","title":"Monte-Carlo Planning: Theoretically Fast Convergence Meets Practical Efficiency","date":"2013-09-26","arxiv_id":"1309.6828","n_code_links":0,"syntology":null},{"paper":null,"slug":"geiringer-theorems-from-population-genetics","title":"Geiringer Theorems: From Population Genetics to Computational Intelligence, Memory Evolutive Systems and Hebbian Learning","date":"2013-05-11","arxiv_id":"1305.2504","n_code_links":0,"syntology":null},{"paper":null,"slug":"monte-carlo-planning-in-large-pomdps","title":"Monte-Carlo Planning in Large POMDPs","date":"2010-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"feature-selection-as-a-one-player-game","title":"Feature Selection as a One-Player Game","date":"2010-05-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/a-monte-carlo-aixi-approximation","slug":"a-monte-carlo-aixi-approximation","title":"A Monte Carlo AIXI Approximation","date":"2009-09-04","arxiv_id":"0909.0801","n_code_links":2,"syntology":null}],"record_sha256":"801861e41c5de75783ddb10476eaadc1332c37b9ff837ba1810d733bfb2b5006","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}