{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/multi-armed-bandits/papers/2","list_of":"/task/multi-armed-bandits","task":"Multi-Armed Bandits","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":13,"rows_per_page":100,"rows":[101,200],"of":1262,"counts":{"archive_papers_tagged":1262,"with_a_code_link":253,"where_syntology_ran_a_sample":55,"not_listed_spam_title":0,"listed":1262,"listed_where_code_ran":55,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":44,"every_run_a_failure_of_syntologys_instrument":11,"listed_with_a_run_with_no_instrument_failure":44,"listed_every_run_a_failure_of_syntologys_instrument":11,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/multi-armed-bandits","prev":"/task/multi-armed-bandits","next":"/task/multi-armed-bandits/papers/3","papers":[{"url":"/paper/implicitly-normalized-forecaster-with","slug":"implicitly-normalized-forecaster-with","title":"Implicitly normalized forecaster with clipping for linear and non-linear heavy-tailed multi-armed bandits","date":"2023-05-11","arxiv_id":"2305.06743","repositories_listed":1,"syntology":null},{"url":"/paper/neural-exploitation-and-exploration-of","slug":"neural-exploitation-and-exploration-of","title":"Neural Exploitation and Exploration of Contextual Bandits","date":"2023-05-05","arxiv_id":"2305.03784","repositories_listed":1,"syntology":null},{"url":"/paper/kullback-leibler-maillard-sampling-for-multi","slug":"kullback-leibler-maillard-sampling-for-multi","title":"Kullback-Leibler Maillard Sampling for Multi-armed Bandits with Bounded Rewards","date":"2023-04-28","arxiv_id":"2304.14989","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/kullback-leibler-maillard-sampling-for-multi#ran","syntology_url":"https://syntology.ai/paper/2304.14989","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.14989"}},"official":{"repos":["MjolnirT/Kullback-Leibler-Maillard-Sampling"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/quantum-natural-policy-gradients-towards","slug":"quantum-natural-policy-gradients-towards","title":"Quantum Natural Policy Gradients: Towards Sample-Efficient Reinforcement Learning","date":"2023-04-26","arxiv_id":"2304.13571","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quantum-natural-policy-gradients-towards#ran","syntology_url":"https://syntology.ai/paper/2304.13571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.13571"}},"official":{"repos":["nicomeyer96/quantum-natural-policy-gradients"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-field-test-of-bandit-algorithms-for","slug":"a-field-test-of-bandit-algorithms-for","title":"A Field Test of Bandit Algorithms for Recommendations: Understanding the Validity of Assumptions on Human Preferences in Multi-armed Bandits","date":"2023-04-16","arxiv_id":"2304.09088","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-evaluation-of-federated","slug":"an-empirical-evaluation-of-federated","title":"An Empirical Evaluation of Federated Contextual Bandit Algorithms","date":"2023-03-17","arxiv_id":"2303.10218","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-empirical-evaluation-of-federated#ran","syntology_url":"https://syntology.ai/paper/2303.10218","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.10218"}},"official":{"repos":["google-research/federated"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/communication-efficient-collaborative","slug":"communication-efficient-collaborative","title":"Flooding with Absorption: An Efficient Protocol for Heterogeneous Bandits over Complex Networks","date":"2023-03-09","arxiv_id":"2303.05445","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-explorative-key-term-selection","slug":"efficient-explorative-key-term-selection","title":"Efficient Explorative Key-term Selection Strategies for Conversational Contextual Bandits","date":"2023-03-01","arxiv_id":"2303.00315","repositories_listed":1,"syntology":null},{"url":"/paper/infinite-action-contextual-bandits-with","slug":"infinite-action-contextual-bandits-with","title":"Infinite Action Contextual Bandits with Reusable Data Exhaust","date":"2023-02-16","arxiv_id":"2302.08551","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/infinite-action-contextual-bandits-with#ran","syntology_url":"https://syntology.ai/paper/2302.08551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.08551"}},"official":{"repos":["mrucker/onoff_experiments"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/piecewise-stationary-multi-objective-multi","slug":"piecewise-stationary-multi-objective-multi","title":"Piecewise-Stationary Multi-Objective Multi-Armed Bandit with Application to Joint Communications and Sensing","date":"2023-02-10","arxiv_id":"2302.05257","repositories_listed":1,"syntology":null},{"url":"/paper/mabsplit-faster-forest-training-using-multi","slug":"mabsplit-faster-forest-training-using-multi","title":"MABSplit: Faster Forest Training Using Multi-Armed Bandits","date":"2022-12-14","arxiv_id":"2212.07473","repositories_listed":1,"syntology":{"n":14,"n_ran":7,"n_constructed":1,"n_ran_checked":2,"n_instrument":5,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":14,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/mabsplit-faster-forest-training-using-multi#ran","syntology_url":"https://syntology.ai/paper/2212.07473","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.07473"}},"official":{"repos":["thrungroup/fastforest"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/networked-restless-bandits-with-positive","slug":"networked-restless-bandits-with-positive","title":"Networked Restless Bandits with Positive Externalities","date":"2022-12-09","arxiv_id":"2212.05144","repositories_listed":1,"syntology":null},{"url":"/paper/stochastic-rising-bandits","slug":"stochastic-rising-bandits","title":"Stochastic Rising Bandits","date":"2022-12-07","arxiv_id":"2212.03798","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stochastic-rising-bandits#ran","syntology_url":"https://syntology.ai/paper/2212.03798","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.03798"}},"official":{"repos":["albertometelli/stochastic-rising-bandits"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ac-band-a-combinatorial-bandit-based-approach","slug":"ac-band-a-combinatorial-bandit-based-approach","title":"AC-Band: A Combinatorial Bandit-Based Approach to Algorithm Configuration","date":"2022-12-01","arxiv_id":"2212.00333","repositories_listed":1,"syntology":null},{"url":"/paper/incorporating-multi-armed-bandit-with-local","slug":"incorporating-multi-armed-bandit-with-local","title":"Incorporating Multi-armed Bandit with Local Search for MaxSAT","date":"2022-11-29","arxiv_id":"2211.16011","repositories_listed":1,"syntology":null},{"url":"/paper/latent-bottlenecked-attentive-neural","slug":"latent-bottlenecked-attentive-neural","title":"Latent Bottlenecked Attentive Neural Processes","date":"2022-11-15","arxiv_id":"2211.08458","repositories_listed":1,"syntology":null},{"url":"/paper/thompson-sampling-for-high-dimensional-sparse","slug":"thompson-sampling-for-high-dimensional-sparse","title":"Thompson Sampling for High-Dimensional Sparse Linear Contextual Bandits","date":"2022-11-11","arxiv_id":"2211.05964","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-real-time-exploration-and","slug":"adaptive-real-time-exploration-and","title":"Safe and Adaptive Decision-Making for Optimization of Safety-Critical Systems: The ARTEO Algorithm","date":"2022-11-10","arxiv_id":"2211.05495","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-data-depth-via-multi-armed-bandits","slug":"adaptive-data-depth-via-multi-armed-bandits","title":"Adaptive Data Depth via Multi-Armed Bandits","date":"2022-11-08","arxiv_id":"2211.03985","repositories_listed":1,"syntology":null},{"url":"/paper/indexability-is-not-enough-for-whittle","slug":"indexability-is-not-enough-for-whittle","title":"Indexability is Not Enough for Whittle: Improved, Near-Optimal Algorithms for Restless Bandits","date":"2022-10-31","arxiv_id":"2211.00112","repositories_listed":1,"syntology":null},{"url":"/paper/conditionally-risk-averse-contextual-bandits","slug":"conditionally-risk-averse-contextual-bandits","title":"Conditionally Risk-Averse Contextual Bandits","date":"2022-10-24","arxiv_id":"2210.13573","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/conditionally-risk-averse-contextual-bandits#ran","syntology_url":"https://syntology.ai/paper/2210.13573","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13573"}},"official":{"repos":["zwd-ms/risk_averse_cb"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/local-metric-learning-for-off-policy","slug":"local-metric-learning-for-off-policy","title":"Local Metric Learning for Off-Policy Evaluation in Contextual Bandits with Continuous Actions","date":"2022-10-24","arxiv_id":"2210.13373","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/local-metric-learning-for-off-policy#ran","syntology_url":"https://syntology.ai/paper/2210.13373","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13373"}},"official":{"repos":["haanvid/kmis"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/fast-beam-alignment-via-pure-exploration-in","slug":"fast-beam-alignment-via-pure-exploration-in","title":"Fast Beam Alignment via Pure Exploration in Multi-armed Bandits","date":"2022-10-23","arxiv_id":"2210.12625","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-contextual-bandits-with-knapsacks","slug":"optimal-contextual-bandits-with-knapsacks","title":"Optimal Contextual Bandits with Knapsacks under Realizability via Regression Oracles","date":"2022-10-21","arxiv_id":"2210.11834","repositories_listed":1,"syntology":null},{"url":"/paper/anytime-valid-off-policy-inference-for","slug":"anytime-valid-off-policy-inference-for","title":"Anytime-valid off-policy inference for contextual bandits","date":"2022-10-19","arxiv_id":"2210.10768","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-dynamic-algorithm-configuration","slug":"multi-agent-dynamic-algorithm-configuration","title":"Multi-agent Dynamic Algorithm Configuration","date":"2022-10-13","arxiv_id":"2210.06835","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-agent-dynamic-algorithm-configuration#ran","syntology_url":"https://syntology.ai/paper/2210.06835","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06835"}},"official":{"repos":["lamda-bbo/madac"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/simulated-contextual-bandits-for","slug":"simulated-contextual-bandits-for","title":"Simulated Contextual Bandits for Personalization Tasks from Recommendation Datasets","date":"2022-10-12","arxiv_id":"2210.10631","repositories_listed":1,"syntology":null},{"url":"/paper/learning-the-trading-algorithm-in-simulated","slug":"learning-the-trading-algorithm-in-simulated","title":"Nonstationary Continuum-Armed Bandit Strategies for Automated Trading in a Simulated Financial Market","date":"2022-08-04","arxiv_id":"2208.02901","repositories_listed":1,"syntology":null},{"url":"/paper/contextual-bandits-with-large-action-spaces","slug":"contextual-bandits-with-large-action-spaces","title":"Contextual Bandits with Large Action Spaces: Made Practical","date":"2022-07-12","arxiv_id":"2207.05836","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/contextual-bandits-with-large-action-spaces#ran","syntology_url":"https://syntology.ai/paper/2207.05836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05836"}},"official":{"repos":["pmineiro/linrepcb"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/contextual-bandits-with-smooth-regret","slug":"contextual-bandits-with-smooth-regret","title":"Contextual Bandits with Smooth Regret: Efficient Learning in Continuous Action Spaces","date":"2022-07-12","arxiv_id":"2207.05849","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/contextual-bandits-with-smooth-regret#ran","syntology_url":"https://syntology.ai/paper/2207.05849","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05849"}},"official":{"repos":["pmineiro/smoothcb"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/transformer-neural-processes-uncertainty","slug":"transformer-neural-processes-uncertainty","title":"Transformer Neural Processes: Uncertainty-Aware Meta Learning Via Sequence Modeling","date":"2022-07-09","arxiv_id":"2207.04179","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/transformer-neural-processes-uncertainty#ran","syntology_url":"https://syntology.ai/paper/2207.04179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.04179"}},"official":{"repos":["tung-nd/tnp-pytorch"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/interactive-combinatorial-bandits-balancing","slug":"interactive-combinatorial-bandits-balancing","title":"Online SuBmodular + SuPermodular (BP) Maximization with Bandit Feedback","date":"2022-07-07","arxiv_id":"2207.03091","repositories_listed":1,"syntology":null},{"url":"/paper/ranking-in-contextual-multi-armed-bandits","slug":"ranking-in-contextual-multi-armed-bandits","title":"Ranking In Generalized Linear Bandits","date":"2022-06-30","arxiv_id":"2207.00109","repositories_listed":1,"syntology":null},{"url":"/paper/two-stage-neural-contextual-bandits-for","slug":"two-stage-neural-contextual-bandits-for","title":"Two-Stage Neural Contextual Bandits for Personalised News Recommendation","date":"2022-06-26","arxiv_id":"2206.14648","repositories_listed":1,"syntology":null},{"url":"/paper/langevin-monte-carlo-for-contextual-bandits","slug":"langevin-monte-carlo-for-contextual-bandits","title":"Langevin Monte Carlo for Contextual Bandits","date":"2022-06-22","arxiv_id":"2206.11254","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/langevin-monte-carlo-for-contextual-bandits#ran","syntology_url":"https://syntology.ai/paper/2206.11254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.11254"}},"official":{"repos":["devzhk/lmcts"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-private-online-convex-optimization-optimal","slug":"on-private-online-convex-optimization-optimal","title":"On Private Online Convex Optimization: Optimal Algorithms in $\\ell_p$-Geometry and High Dimensional Contextual Bandits","date":"2022-06-16","arxiv_id":"2206.08111","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":3,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/on-private-online-convex-optimization-optimal#ran","syntology_url":"https://syntology.ai/paper/2206.08111","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08111"}},"official":{"repos":["liangzp/dp-streaming-sco"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/group-meritocratic-fairness-in-linear","slug":"group-meritocratic-fairness-in-linear","title":"Group Meritocratic Fairness in Linear Contextual Bandits","date":"2022-06-07","arxiv_id":"2206.03150","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/group-meritocratic-fairness-in-linear#ran","syntology_url":"https://syntology.ai/paper/2206.03150","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.03150"}},"official":{"repos":["csml-iit-ucl/gmfbandits"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/optimistic-whittle-index-policy-online","slug":"optimistic-whittle-index-policy-online","title":"Optimistic Whittle Index Policy: Online Learning for Restless Bandits","date":"2022-05-30","arxiv_id":"2205.15372","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimistic-whittle-index-policy-online#ran","syntology_url":"https://syntology.ai/paper/2205.15372","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.15372"}},"official":{"repos":["lily-x/online-rmab"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/federated-neural-bandit","slug":"federated-neural-bandit","title":"Federated Neural Bandits","date":"2022-05-28","arxiv_id":"2205.14309","repositories_listed":1,"syntology":null},{"url":"/paper/optimality-conditions-and-algorithms-for-top","slug":"optimality-conditions-and-algorithms-for-top","title":"Information-Directed Selection for Top-Two Algorithms","date":"2022-05-24","arxiv_id":"2205.12086","repositories_listed":1,"syntology":null},{"url":"/paper/splitplace-ai-augmented-splitting-and","slug":"splitplace-ai-augmented-splitting-and","title":"SplitPlace: AI Augmented Splitting and Placement of Large-Scale Neural Networks in Mobile Edge Environments","date":"2022-05-21","arxiv_id":"2205.10635","repositories_listed":1,"syntology":null},{"url":"/paper/multi-armed-bandits-in-brain-computer","slug":"multi-armed-bandits-in-brain-computer","title":"Multi-Armed Bandits in Brain-Computer Interfaces","date":"2022-05-19","arxiv_id":"2205.09584","repositories_listed":1,"syntology":null},{"url":"/paper/pervasive-machine-learning-for-smart-radio","slug":"pervasive-machine-learning-for-smart-radio","title":"Pervasive Machine Learning for Smart Radio Environments Enabled by Reconfigurable Intelligent Surfaces","date":"2022-05-08","arxiv_id":"2205.03793","repositories_listed":1,"syntology":null},{"url":"/paper/evolutionary-multi-armed-bandits-with-genetic","slug":"evolutionary-multi-armed-bandits-with-genetic","title":"Evolutionary Multi-Armed Bandits with Genetic Thompson Sampling","date":"2022-04-26","arxiv_id":"2205.10113","repositories_listed":1,"syntology":null},{"url":"/paper/thompson-sampling-for-bandit-learning-in","slug":"thompson-sampling-for-bandit-learning-in","title":"Thompson Sampling for Bandit Learning in Matching Markets","date":"2022-04-26","arxiv_id":"2204.12048","repositories_listed":1,"syntology":null},{"url":"/paper/multi-armed-bandits-for-online-optimization","slug":"multi-armed-bandits-for-online-optimization","title":"Multi-armed bandits for resource efficient, online optimization of language model pre-training: the use case of dynamic masking","date":"2022-03-24","arxiv_id":"2203.13151","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-algorithms-for-extreme-bandits","slug":"efficient-algorithms-for-extreme-bandits","title":"Efficient Algorithms for Extreme Bandits","date":"2022-03-21","arxiv_id":"2203.10883","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-kernel-ucb-for-contextual-bandits","slug":"efficient-kernel-ucb-for-contextual-bandits","title":"Efficient Kernel UCB for Contextual Bandits","date":"2022-02-11","arxiv_id":"2202.05638","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-experimentation-with-delayed-binary","slug":"adaptive-experimentation-with-delayed-binary","title":"Adaptive Experimentation with Delayed Binary Feedback","date":"2022-02-02","arxiv_id":"2202.00846","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-deep-vs-wide-deep-learners-as","slug":"evaluating-deep-vs-wide-deep-learners-as","title":"Evaluating Deep Vs. Wide & Deep Learners As Contextual Bandits For Personalized Email Promo Recommendations","date":"2022-01-31","arxiv_id":"2202.00146","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-regret-is-achievable-with-bounded-1","slug":"optimal-regret-is-achievable-with-bounded-1","title":"Optimal Regret Is Achievable with Bounded Approximate Inference Error: An Enhanced Bayesian Upper Confidence Bound Framework","date":"2022-01-31","arxiv_id":"2201.12955","repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-evaluation-using-information","slug":"off-policy-evaluation-using-information","title":"Off-Policy Evaluation Using Information Borrowing and Context-Based Switching","date":"2021-12-18","arxiv_id":"2112.09865","repositories_listed":1,"syntology":null},{"url":"/paper/identification-of-the-generalized-condorcet","slug":"identification-of-the-generalized-condorcet","title":"Identification of the Generalized Condorcet Winner in Multi-dueling Bandits","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/subgaussian-and-differentiable-importance","slug":"subgaussian-and-differentiable-importance","title":"Subgaussian and Differentiable Importance Sampling for Off-Policy Evaluation and Learning","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/offline-neural-contextual-bandits-pessimism-1","slug":"offline-neural-contextual-bandits-pessimism-1","title":"Offline Neural Contextual Bandits: Pessimism, Optimization and Generalization","date":"2021-11-27","arxiv_id":"2111.13807","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":3,"n_ran_checked":3,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":8,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/offline-neural-contextual-bandits-pessimism-1#ran","syntology_url":"https://syntology.ai/paper/2111.13807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.13807"}},"official":{"repos":["thanhnguyentang/offline_neural_bandits"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/almost-free-incentivized-exploration-from","slug":"almost-free-incentivized-exploration-from","title":"(Almost) Free Incentivized Exploration from Decentralized Learning Agents","date":"2021-10-27","arxiv_id":"2110.14628","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/almost-free-incentivized-exploration-from#ran","syntology_url":"https://syntology.ai/paper/2110.14628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.14628"}},"official":{"repos":["shengroup/observe_then_incentivize"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/heterogeneous-multi-player-multi-armed","slug":"heterogeneous-multi-player-multi-armed","title":"Heterogeneous Multi-player Multi-armed Bandits: Closing the Gap and Generalization","date":"2021-10-27","arxiv_id":"2110.14622","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/heterogeneous-multi-player-multi-armed#ran","syntology_url":"https://syntology.ai/paper/2110.14622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.14622"}},"official":{"repos":["shengroup/mpmab_beacon"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-the-d-optimal-online-experiment","slug":"towards-the-d-optimal-online-experiment","title":"Towards the D-Optimal Online Experiment Design for Recommender Selection","date":"2021-10-23","arxiv_id":"2110.12132","repositories_listed":1,"syntology":null},{"url":"/paper/ee-net-exploitation-exploration-neural","slug":"ee-net-exploitation-exploration-neural","title":"EE-Net: Exploitation-Exploration Neural Networks in Contextual Bandits","date":"2021-10-07","arxiv_id":"2110.03177","repositories_listed":1,"syntology":null},{"url":"/paper/estimation-of-warfarin-dosage-with","slug":"estimation-of-warfarin-dosage-with","title":"Estimation of Warfarin Dosage with Reinforcement Learning","date":"2021-09-15","arxiv_id":"2109.07564","repositories_listed":1,"syntology":null},{"url":"/paper/pure-exploration-in-multi-armed-bandits-with","slug":"pure-exploration-in-multi-armed-bandits-with","title":"Maximizing and Satisficing in Multi-armed Bandits with Graph Information","date":"2021-08-02","arxiv_id":"2108.01152","repositories_listed":1,"syntology":null},{"url":"/paper/finite-time-analysis-of-globally","slug":"finite-time-analysis-of-globally","title":"Finite-time Analysis of Globally Nonstationary Multi-Armed Bandits","date":"2021-07-23","arxiv_id":"2107.11419","repositories_listed":1,"syntology":null},{"url":"/paper/neural-contextual-bandits-without-regret","slug":"neural-contextual-bandits-without-regret","title":"Neural Contextual Bandits without Regret","date":"2021-07-07","arxiv_id":"2107.03144","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/neural-contextual-bandits-without-regret#ran","syntology_url":"https://syntology.ai/paper/2107.03144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.03144"}},"official":{"repos":["pkassraie/NNUCB"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/q-learning-lagrange-policies-for-multi-action","slug":"q-learning-lagrange-policies-for-multi-action","title":"Q-Learning Lagrange Policies for Multi-Action Restless Bandits","date":"2021-06-22","arxiv_id":"2106.12024","repositories_listed":1,"syntology":{"n":17,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/q-learning-lagrange-policies-for-multi-action#ran","syntology_url":"https://syntology.ai/paper/2106.12024","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.12024"}},"official":{"repos":["killian-34/MAIQL_and_LPQL"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-phy-layer","slug":"reinforcement-learning-for-phy-layer","title":"Reinforcement Learning for Physical Layer Communications","date":"2021-06-22","arxiv_id":"2106.11595","repositories_listed":1,"syntology":null},{"url":"/paper/generalized-linear-bandits-with-local","slug":"generalized-linear-bandits-with-local","title":"Generalized Linear Bandits with Local Differential Privacy","date":"2021-06-07","arxiv_id":"2106.03365","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/generalized-linear-bandits-with-local#ran","syntology_url":"https://syntology.ai/paper/2106.03365","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03365"}},"official":{"repos":["liangzp/LDP-Bandit"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/multi-facet-contextual-bandits-a-neural","slug":"multi-facet-contextual-bandits-a-neural","title":"Multi-facet Contextual Bandits: A Neural Network Perspective","date":"2021-06-06","arxiv_id":"2106.03039","repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-evaluation-via-adaptive-weighting","slug":"off-policy-evaluation-via-adaptive-weighting","title":"Off-Policy Evaluation via Adaptive Weighting with Data from Contextual Bandits","date":"2021-06-03","arxiv_id":"2106.02029","repositories_listed":1,"syntology":{"n":25,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":12,"n_honours":0,"n_violates":1,"n_no_contract":12,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 1 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 12 unverified","sample_list":"/paper/off-policy-evaluation-via-adaptive-weighting#ran","syntology_url":"https://syntology.ai/paper/2106.02029","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.02029"}},"official":{"repos":["gsbDBI/contextual_bandits_evaluation"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":12,"ran_from_kinds":["official"]}}},{"url":"/paper/addressing-the-long-term-impact-of-ml","slug":"addressing-the-long-term-impact-of-ml","title":"Addressing the Long-term Impact of ML Decisions via Policy Regret","date":"2021-06-02","arxiv_id":"2106.01325","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/addressing-the-long-term-impact-of-ml#ran","syntology_url":"https://syntology.ai/paper/2106.01325","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.01325"}},"official":{"repos":["david-lindner/single-peaked-bandits"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/invariant-policy-learning-a-causal","slug":"invariant-policy-learning-a-causal","title":"Invariant Policy Learning: A Causal Perspective","date":"2021-06-01","arxiv_id":"2106.00808","repositories_listed":1,"syntology":null},{"url":"/paper/combinatorial-multi-armed-bandits-for","slug":"combinatorial-multi-armed-bandits-for","title":"Combinatorial Multi-armed Bandits for Resource Allocation","date":"2021-05-10","arxiv_id":"2105.04373","repositories_listed":1,"syntology":null},{"url":"/paper/deep-bandits-show-off-simple-and-efficient","slug":"deep-bandits-show-off-simple-and-efficient","title":"Deep Bandits Show-Off: Simple and Efficient Exploration with Deep Networks","date":"2021-05-10","arxiv_id":"2105.04683","repositories_listed":1,"syntology":null},{"url":"/paper/policy-learning-with-adaptively-collected","slug":"policy-learning-with-adaptively-collected","title":"Policy Learning with Adaptively Collected Data","date":"2021-05-05","arxiv_id":"2105.02344","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/policy-learning-with-adaptively-collected#ran","syntology_url":"https://syntology.ai/paper/2105.02344","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.02344"}},"official":{"repos":["gsbDBI/PolicyLearning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/combinatorial-bandits-under-strategic","slug":"combinatorial-bandits-under-strategic","title":"Combinatorial Bandits under Strategic Manipulations","date":"2021-02-25","arxiv_id":"2102.12722","repositories_listed":1,"syntology":null},{"url":"/paper/federated-multi-armed-bandits-with","slug":"federated-multi-armed-bandits-with","title":"Federated Multi-armed Bandits with Personalization","date":"2021-02-25","arxiv_id":"2102.13101","repositories_listed":1,"syntology":null},{"url":"/paper/output-weighted-sampling-for-multi-armed","slug":"output-weighted-sampling-for-multi-armed","title":"Output-Weighted Sampling for Multi-Armed Bandits with Extreme Payoffs","date":"2021-02-19","arxiv_id":"2102.10085","repositories_listed":1,"syntology":null},{"url":"/paper/top-k-extreme-contextual-bandits-with-arm","slug":"top-k-extreme-contextual-bandits-with-arm","title":"Top-$k$ eXtreme Contextual Bandits with Arm Hierarchy","date":"2021-02-15","arxiv_id":"2102.07800","repositories_listed":1,"syntology":null},{"url":"/paper/federated-multi-armed-bandits","slug":"federated-multi-armed-bandits","title":"Federated Multi-Armed Bandits","date":"2021-01-28","arxiv_id":"2101.12204","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-evaluation-of-active-inference","slug":"an-empirical-evaluation-of-active-inference","title":"An empirical evaluation of active inference in multi-armed bandits","date":"2021-01-21","arxiv_id":"2101.08699","repositories_listed":1,"syntology":null},{"url":"/paper/relational-boosted-bandits","slug":"relational-boosted-bandits","title":"Relational Boosted Bandits","date":"2020-12-16","arxiv_id":"2012.09220","repositories_listed":1,"syntology":null},{"url":"/paper/active-feature-selection-for-the-mutual","slug":"active-feature-selection-for-the-mutual","title":"Active Feature Selection for the Mutual Information Criterion","date":"2020-12-13","arxiv_id":"2012.06979","repositories_listed":1,"syntology":null},{"url":"/paper/banditpam-almost-linear-time-k-medoids","slug":"banditpam-almost-linear-time-k-medoids","title":"BanditPAM: Almost Linear Time k-Medoids Clustering via Multi-Armed Bandits","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/unreasonable-effectiveness-of-greedy","slug":"unreasonable-effectiveness-of-greedy","title":"Unreasonable Effectiveness of Greedy Algorithms in Multi-Armed Bandit with Many Arms","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/corrupted-contextual-bandits-with-action","slug":"corrupted-contextual-bandits-with-action","title":"A New Bandit Setting Balancing Information from State Evolution and Corrupted Context","date":"2020-11-16","arxiv_id":"2011.07989","repositories_listed":1,"syntology":null},{"url":"/paper/multi-armed-bandits-with-cost-subsidy","slug":"multi-armed-bandits-with-cost-subsidy","title":"Multi-armed Bandits with Cost Subsidy","date":"2020-11-03","arxiv_id":"2011.01488","repositories_listed":1,"syntology":null},{"url":"/paper/online-semi-supervised-learning-in-contextual","slug":"online-semi-supervised-learning-in-contextual","title":"Online Semi-Supervised Learning in Contextual Bandits with Episodic Reward","date":"2020-09-17","arxiv_id":"2009.08457","repositories_listed":1,"syntology":null},{"url":"/paper/carousel-personalization-in-music-streaming","slug":"carousel-personalization-in-music-streaming","title":"Carousel Personalization in Music Streaming Apps with Contextual Bandits","date":"2020-09-14","arxiv_id":"2009.06546","repositories_listed":1,"syntology":null},{"url":"/paper/vacsim-learning-effective-strategies-for","slug":"vacsim-learning-effective-strategies-for","title":"VacSIM: Learning Effective Strategies for COVID-19 Vaccine Distribution using Reinforcement Learning","date":"2020-09-14","arxiv_id":"2009.06602","repositories_listed":1,"syntology":null},{"url":"/paper/using-subjective-logic-to-estimate","slug":"using-subjective-logic-to-estimate","title":"Using Subjective Logic to Estimate Uncertainty in Multi-Armed Bandit Problems","date":"2020-08-17","arxiv_id":"2008.07386","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-learning-for-structured-bandits","slug":"optimal-learning-for-structured-bandits","title":"Optimal Learning for Structured Bandits","date":"2020-07-14","arxiv_id":"2007.07302","repositories_listed":1,"syntology":null},{"url":"/paper/quantum-exploration-algorithms-for-multi","slug":"quantum-exploration-algorithms-for-multi","title":"Quantum exploration algorithms for multi-armed bandits","date":"2020-07-14","arxiv_id":"2007.07049","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"0 ran · 3 unverified","sample_list":"/paper/quantum-exploration-algorithms-for-multi#ran","syntology_url":"https://syntology.ai/paper/2007.07049","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.07049"}},"official":null}},{"url":"/paper/recurrent-neural-linear-posterior-sampling","slug":"recurrent-neural-linear-posterior-sampling","title":"Recurrent Neural-Linear Posterior Sampling for Nonstationary Contextual Bandits","date":"2020-07-09","arxiv_id":"2007.04750","repositories_listed":1,"syntology":null},{"url":"/paper/overfitting-and-optimization-in-offline","slug":"overfitting-and-optimization-in-offline","title":"Offline Contextual Bandits with Overparameterized Models","date":"2020-06-27","arxiv_id":"2006.15368","repositories_listed":1,"syntology":null},{"url":"/paper/constrained-regret-minimization-for-multi","slug":"constrained-regret-minimization-for-multi","title":"Constrained regret minimization for multi-criterion multi-armed bandits","date":"2020-06-17","arxiv_id":"2006.09649","repositories_listed":1,"syntology":null},{"url":"/paper/finding-all-e-good-arms-in-stochastic-bandits","slug":"finding-all-e-good-arms-in-stochastic-bandits","title":"Finding All ε-Good Arms in Stochastic Bandits","date":"2020-06-16","arxiv_id":"2006.08850","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-contextual-bandits-with-continuous","slug":"efficient-contextual-bandits-with-continuous","title":"Efficient Contextual Bandits with Continuous Actions","date":"2020-06-10","arxiv_id":"2006.06040","repositories_listed":1,"syntology":null},{"url":"/paper/online-learning-in-iterated-prisoner-s","slug":"online-learning-in-iterated-prisoner-s","title":"Online Learning in Iterated Prisoner's Dilemma to Mimic Human Behavior","date":"2020-06-09","arxiv_id":"2006.06580","repositories_listed":1,"syntology":null},{"url":"/paper/unified-models-of-human-behavioral-agents-in","slug":"unified-models-of-human-behavioral-agents-in","title":"Unified Models of Human Behavioral Agents in Bandits, Contextual Bandits and RL","date":"2020-05-10","arxiv_id":"2005.04544","repositories_listed":1,"syntology":null},{"url":"/paper/thompson-sampling-for-linearly-constrained","slug":"thompson-sampling-for-linearly-constrained","title":"Thompson Sampling for Linearly Constrained Bandits","date":"2020-04-20","arxiv_id":"2004.09258","repositories_listed":1,"syntology":null},{"url":"/paper/power-constrained-bandits","slug":"power-constrained-bandits","title":"Power Constrained Bandits","date":"2020-04-13","arxiv_id":"2004.06230","repositories_listed":1,"syntology":null}],"record_sha256":"c4237f15c3cc0cf3cde1ad57c2fc3d7beb71f0dc70a26fd872c76bc10b7e991b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}