{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/thompson-sampling/papers/7","list_of":"/task/thompson-sampling","task":"Thompson Sampling","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":7,"pages_in_order":7,"rows_per_page":100,"rows":[601,655],"of":655,"counts":{"archive_papers_tagged":655,"with_a_code_link":135,"where_syntology_ran_a_sample":30,"not_listed_spam_title":0,"listed":655,"listed_where_code_ran":30,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":23,"every_run_a_failure_of_syntologys_instrument":7,"listed_with_a_run_with_no_instrument_failure":23,"listed_every_run_a_failure_of_syntologys_instrument":7,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/thompson-sampling","prev":"/task/thompson-sampling/papers/6","next":null,"papers":[{"url":null,"slug":"analysis-of-thompson-sampling-for-gaussian","title":"Adaptive Rate of Convergence of Thompson Sampling for Gaussian Process Optimization","date":"2017-05-18","arxiv_id":"1705.06808","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-attentive-bandits-contextual-bandit","title":"Context Attentive Bandits: Contextual Bandit with Restricted Context","date":"2017-05-10","arxiv_id":"1705.03821","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-dueling-bandits-with-dependent-arms","title":"Multi-dueling Bandits with Dependent Arms","date":"2017-04-29","arxiv_id":"1705.00253","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-sensitive-bandit-learning-and","title":"Time-Sensitive Bandit Learning and Satisficing Thompson Sampling","date":"2017-04-28","arxiv_id":"1704.09028","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-benchmarking-of-nlp-apis-using","title":"Efficient Benchmarking of NLP APIs using Multi-armed Bandits","date":"2017-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"thompson-sampling-for-linear-quadratic","title":"Thompson Sampling for Linear-Quadratic Control Problems","date":"2017-03-27","arxiv_id":"1703.08972","repositories_listed":0,"syntology":null},{"url":null,"slug":"horde-of-bandits-using-gaussian-markov-random","title":"Horde of Bandits using Gaussian Markov Random Fields","date":"2017-03-07","arxiv_id":"1703.02626","repositories_listed":0,"syntology":null},{"url":null,"slug":"qos-aware-multi-armed-bandits","title":"QoS-Aware Multi-Armed Bandits","date":"2017-02-28","arxiv_id":"1703.10669","repositories_listed":0,"syntology":null},{"url":null,"slug":"thompson-sampling-for-stochastic-bandits-with","title":"Thompson Sampling For Stochastic Bandits with Graph Feedback","date":"2017-01-16","arxiv_id":"1701.04238","repositories_listed":0,"syntology":null},{"url":null,"slug":"estimating-quality-in-multi-objective-bandits","title":"Estimating Quality in Multi-Objective Bandits Optimization","date":"2017-01-04","arxiv_id":"1701.01095","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-for-multi-task-reinforcement","title":"Exploration for Multi-task Reinforcement Learning with Deep Generative Models","date":"2016-11-29","arxiv_id":"1611.09894","repositories_listed":0,"syntology":null},{"url":null,"slug":"nonparametric-general-reinforcement-learning","title":"Nonparametric General Reinforcement Learning","date":"2016-11-28","arxiv_id":"1611.08944","repositories_listed":0,"syntology":null},{"url":null,"slug":"linear-thompson-sampling-revisited","title":"Linear Thompson Sampling Revisited","date":"2016-11-20","arxiv_id":"1611.06534","repositories_listed":0,"syntology":null},{"url":null,"slug":"unimodal-thompson-sampling-for-graph","title":"Unimodal Thompson Sampling for Graph-Structured Arms","date":"2016-11-17","arxiv_id":"1611.05724","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-end-of-optimism-an-asymptotic-analysis-of","title":"The End of Optimism? An Asymptotic Analysis of Finite-Armed Linear Bandits","date":"2016-10-14","arxiv_id":"1610.04491","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-formal-solution-to-the-grain-of-truth","title":"A Formal Solution to the Grain of Truth Problem","date":"2016-09-16","arxiv_id":"1609.05058","repositories_listed":0,"syntology":null},{"url":null,"slug":"bbq-networks-efficient-exploration-in-deep-1","title":"BBQ-Networks: Efficient Exploration in Deep Reinforcement Learning for Task-Oriented Dialogue Systems","date":"2016-08-17","arxiv_id":"1608.05081","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-collective-intelligence-as-distributed","title":"Human collective intelligence as distributed Bayesian inference","date":"2016-08-05","arxiv_id":"1608.01987","repositories_listed":0,"syntology":null},{"url":null,"slug":"asymptotically-optimal-algorithms-for","title":"Asymptotically Optimal Algorithms for Budgeted Multiple Play Bandits","date":"2016-06-30","arxiv_id":"1606.09388","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-algorithms-for-parameter-mean-and","title":"Online Algorithms For Parameter Mean And Variance Estimation In Dynamic Regression Models","date":"2016-05-18","arxiv_id":"1605.05697","repositories_listed":0,"syntology":null},{"url":null,"slug":"linear-bandit-algorithms-using-the-bootstrap","title":"Linear Bandit algorithms using the Bootstrap","date":"2016-05-04","arxiv_id":"1605.01185","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-unbiased-data-collection-and-content","title":"An Unbiased Data Collection and Content Exploitation/Exploration Strategy for Personalization","date":"2016-04-12","arxiv_id":"1604.03506","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-sequential-monte-carlo-approach-to-thompson","title":"A sequential Monte Carlo approach to Thompson sampling for Bayesian optimization","date":"2016-04-01","arxiv_id":"1604.00169","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-recommendation-to-users-that-react","title":"Optimal Recommendation to Users that React: Online Learning for a Class of POMDPs","date":"2016-03-30","arxiv_id":"1603.09233","repositories_listed":0,"syntology":null},{"url":null,"slug":"thompson-sampling-is-asymptotically-optimal","title":"Thompson Sampling is Asymptotically Optimal in General Environments","date":"2016-02-25","arxiv_id":"1602.07905","repositories_listed":0,"syntology":null},{"url":null,"slug":"convolutional-monte-carlo-rollouts-in-go","title":"Convolutional Monte Carlo Rollouts in Go","date":"2015-12-10","arxiv_id":"1512.03375","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-thompson-sampling-for-online-matrix","title":"Efficient Thompson Sampling for Online ￼Matrix-Factorization Recommendation","date":"2015-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"regret-analysis-of-the-finite-horizon-gittins","title":"Regret Analysis of the Finite-Horizon Gittins Index Strategy for Multi-Armed Bandits","date":"2015-11-18","arxiv_id":"1511.06014","repositories_listed":0,"syntology":null},{"url":null,"slug":"tseb-more-efficient-thompson-sampling-for","title":"TSEB: More Efficient Thompson Sampling for Policy Learning","date":"2015-10-10","arxiv_id":"1510.02874","repositories_listed":0,"syntology":null},{"url":null,"slug":"bootstrapped-thompson-sampling-and-deep","title":"Bootstrapped Thompson Sampling and Deep Exploration","date":"2015-07-01","arxiv_id":"1507.00300","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-prior-sensitivity-of-thompson-sampling","title":"On the Prior Sensitivity of Thompson Sampling","date":"2015-06-10","arxiv_id":"1506.03378","repositories_listed":0,"syntology":null},{"url":null,"slug":"belief-flows-of-robust-online-learning","title":"Belief Flows of Robust Online Learning","date":"2015-05-26","arxiv_id":"1505.07067","repositories_listed":0,"syntology":null},{"url":null,"slug":"thompson-sampling-for-budgeted-multi-armed","title":"Thompson Sampling for Budgeted Multi-armed Bandits","date":"2015-05-01","arxiv_id":"1505.00146","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-explore-exploit-policies-in","title":"Evaluation of Explore-Exploit Policies in Multi-result Ranking Systems","date":"2015-04-28","arxiv_id":"1504.07662","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-note-on-information-directed-sampling-and","title":"A Note on Information-Directed Sampling and Thompson Sampling","date":"2015-03-24","arxiv_id":"1503.06902","repositories_listed":0,"syntology":null},{"url":null,"slug":"bandit-convex-optimization-sqrtt-regret-in","title":"Bandit Convex Optimization: sqrt{T} Regret in One Dimension","date":"2015-02-23","arxiv_id":"1502.06398","repositories_listed":0,"syntology":null},{"url":null,"slug":"thompson-sampling-with-the-online-bootstrap","title":"Thompson sampling with the online bootstrap","date":"2014-10-15","arxiv_id":"1410.4009","repositories_listed":0,"syntology":null},{"url":null,"slug":"freshness-aware-thompson-sampling","title":"Freshness-Aware Thompson Sampling","date":"2014-09-29","arxiv_id":"1409.8572","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-optimal-algorithms-for-prediction","title":"Towards Optimal Algorithms for Prediction with Expert Advice","date":"2014-09-10","arxiv_id":"1409.3040","repositories_listed":0,"syntology":null},{"url":null,"slug":"thompson-sampling-for-learning-parameterized","title":"Thompson Sampling for Learning Parameterized Markov Decision Processes","date":"2014-06-29","arxiv_id":"1406.7498","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-learning-in-large-scale","title":"Efficient Learning in Large-Scale Combinatorial Semi-Bandits","date":"2014-06-28","arxiv_id":"1406.7443","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-information-theoretic-analysis-of-thompson","title":"An Information-Theoretic Analysis of Thompson Sampling","date":"2014-03-21","arxiv_id":"1403.5341","repositories_listed":0,"syntology":null},{"url":null,"slug":"better-optimism-by-bayes-adaptive-planning","title":"Better Optimism By Bayes: Adaptive Planning with Rich Models","date":"2014-02-09","arxiv_id":"1402.1958","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-mixture-modelling-and-inference","title":"Bayesian Mixture Modelling and Inference based Thompson Sampling in Monte-Carlo Tree Search","date":"2013-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"eluder-dimension-and-the-sample-complexity-of","title":"Eluder Dimension and the Sample Complexity of Optimistic Exploration","date":"2013-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"thompson-sampling-for-complex-bandit-problems","title":"Thompson Sampling for Complex Bandit Problems","date":"2013-11-03","arxiv_id":"1311.0466","repositories_listed":0,"syntology":null},{"url":null,"slug":"thompson-sampling-for-online-learning-with","title":"Thompson Sampling for Online Learning with Linear Experts","date":"2013-11-03","arxiv_id":"1311.0468","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-thompson-sampling-for-contextual","title":"Generalized Thompson Sampling for Contextual Bandits","date":"2013-10-27","arxiv_id":"1310.7163","repositories_listed":0,"syntology":null},{"url":null,"slug":"thompson-sampling-in-dynamic-systems-for","title":"Thompson Sampling in Dynamic Systems for Contextual Bandit Problems","date":"2013-10-17","arxiv_id":"1310.5008","repositories_listed":0,"syntology":null},{"url":null,"slug":"thompson-sampling-for-1-dimensional","title":"Thompson Sampling for 1-Dimensional Exponential Family Bandits","date":"2013-07-12","arxiv_id":"1307.3400","repositories_listed":0,"syntology":null},{"url":null,"slug":"cover-tree-bayesian-reinforcement-learning","title":"Cover Tree Bayesian Reinforcement Learning","date":"2013-05-08","arxiv_id":"1305.1809","repositories_listed":0,"syntology":null},{"url":null,"slug":"prior-free-and-prior-dependent-regret-bounds","title":"Prior-free and prior-dependent regret bounds for Thompson Sampling","date":"2013-04-21","arxiv_id":"1304.5758","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-correlation-and-budget-constraints","title":"Exploiting correlation and budget constraints in Bayesian multi-armed bandit optimization","date":"2013-03-27","arxiv_id":"1303.6746","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-optimize-via-posterior-sampling","title":"Learning to Optimize Via Posterior Sampling","date":"2013-01-11","arxiv_id":"1301.2609","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-empirical-evaluation-of-thompson-sampling","title":"An Empirical Evaluation of Thompson Sampling","date":"2011-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"c2a8080772685d6bce9a833e179e0cde265feabc5a7cfc993f4cc3184b9848eb","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}