{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/sequential-decision-making/papers/3","list_of":"/task/sequential-decision-making","task":"Sequential Decision Making","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":13,"rows_per_page":100,"rows":[201,300],"of":1210,"counts":{"archive_papers_tagged":1210,"with_a_code_link":351,"where_syntology_ran_a_sample":107,"not_listed_spam_title":0,"listed":1210,"listed_where_code_ran":107,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":90,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":90,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/sequential-decision-making","prev":"/task/sequential-decision-making/papers/2","next":"/task/sequential-decision-making/papers/4","papers":[{"url":"/paper/scalable-bayesian-optimization-with-high","slug":"scalable-bayesian-optimization-with-high","title":"Scalable Bayesian optimization with high-dimensional outputs using randomized prior networks","date":"2023-02-14","arxiv_id":"2302.07260","repositories_listed":1,"syntology":null},{"url":"/paper/variational-information-pursuit-for","slug":"variational-information-pursuit-for","title":"Variational Information Pursuit for Interpretable Predictions","date":"2023-02-06","arxiv_id":"2302.02876","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/variational-information-pursuit-for#ran","syntology_url":"https://syntology.ai/paper/2302.02876","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.02876"}},"official":{"repos":["ryanchankh/VariationalInformationPursuit"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-coordination-policies-over","slug":"learning-coordination-policies-over","title":"Learning Coordination Policies over Heterogeneous Graphs for Human-Robot Teams via Recurrent Neural Schedule Propagation","date":"2023-01-30","arxiv_id":"2301.13279","repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-evaluation-for-action-dependent","slug":"off-policy-evaluation-for-action-dependent","title":"Off-Policy Evaluation for Action-Dependent Non-Stationary Environments","date":"2023-01-24","arxiv_id":"2301.10330","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/off-policy-evaluation-for-action-dependent#ran","syntology_url":"https://syntology.ai/paper/2301.10330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.10330"}},"official":{"repos":["yashchandak/activens"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-conditional-cauchy-schwarz-divergence","slug":"the-conditional-cauchy-schwarz-divergence","title":"The Conditional Cauchy-Schwarz Divergence with Applications to Time-Series Data and Sequential Decision Making","date":"2023-01-21","arxiv_id":"2301.08970","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-conditional-cauchy-schwarz-divergence#ran","syntology_url":"https://syntology.ai/paper/2301.08970","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.08970"}},"official":{"repos":["sjyucnel/conditional_cs_divergence"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/differential-privacy-in-cooperative","slug":"differential-privacy-in-cooperative","title":"Differential Privacy in Cooperative Multiagent Planning","date":"2023-01-20","arxiv_id":"2301.08811","repositories_listed":1,"syntology":null},{"url":"/paper/plan-to-predict-learning-an-uncertainty","slug":"plan-to-predict-learning-an-uncertainty","title":"Plan To Predict: Learning an Uncertainty-Foreseeing Model for Model-Based Reinforcement Learning","date":"2023-01-20","arxiv_id":"2301.08502","repositories_listed":1,"syntology":null},{"url":"/paper/risk-sensitive-policy-with-distributional","slug":"risk-sensitive-policy-with-distributional","title":"Risk-Sensitive Policy with Distributional Reinforcement Learning","date":"2022-12-30","arxiv_id":"2212.14743","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-pomdps-and-bayesian-decision-making","slug":"bridging-pomdps-and-bayesian-decision-making","title":"Bridging POMDPs and Bayesian decision making for robust maintenance planning under model uncertainty: An application to railway systems","date":"2022-12-15","arxiv_id":"2212.07933","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-multi-agent-deep-reinforcement","slug":"hybrid-multi-agent-deep-reinforcement","title":"Hybrid Multi-agent Deep Reinforcement Learning for Autonomous Mobility on Demand Systems","date":"2022-12-14","arxiv_id":"2212.07313","repositories_listed":1,"syntology":null},{"url":"/paper/autoregressive-bandits","slug":"autoregressive-bandits","title":"Autoregressive Bandits","date":"2022-12-12","arxiv_id":"2212.06251","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autoregressive-bandits#ran","syntology_url":"https://syntology.ai/paper/2212.06251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.06251"}},"official":{"repos":["gianmarcogenalti/autoregressive-bandits"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/information-theoretic-safe-exploration-with","slug":"information-theoretic-safe-exploration-with","title":"Information-Theoretic Safe Exploration with Gaussian Processes","date":"2022-12-09","arxiv_id":"2212.04914","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":2,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":8,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/information-theoretic-safe-exploration-with#ran","syntology_url":"https://syntology.ai/paper/2212.04914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.04914"}},"official":{"repos":["boschresearch/information-theoretic-safe-exploration"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/ace-cooperative-multi-agent-q-learning-with","slug":"ace-cooperative-multi-agent-q-learning-with","title":"ACE: Cooperative Multi-agent Q-learning with Bidirectional Action-Dependency","date":"2022-11-29","arxiv_id":"2211.16068","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ace-cooperative-multi-agent-q-learning-with#ran","syntology_url":"https://syntology.ai/paper/2211.16068","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.16068"}},"official":{"repos":["opendilab/ace"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/anderson-acceleration-for-partially","slug":"anderson-acceleration-for-partially","title":"Anderson Acceleration for Partially Observable Markov Decision Processes: A Maximum Entropy Approach","date":"2022-11-28","arxiv_id":"2211.14998","repositories_listed":1,"syntology":null},{"url":"/paper/b-multivariational-autoencoder-for-entangled","slug":"b-multivariational-autoencoder-for-entangled","title":"$β$-Multivariational Autoencoder for Entangled Representation Learning in Video Frames","date":"2022-11-22","arxiv_id":"2211.12627","repositories_listed":1,"syntology":null},{"url":"/paper/unimask-unified-inference-in-sequential","slug":"unimask-unified-inference-in-sequential","title":"UniMASK: Unified Inference in Sequential Decision Problems","date":"2022-11-20","arxiv_id":"2211.10869","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/unimask-unified-inference-in-sequential#ran","syntology_url":"https://syntology.ai/paper/2211.10869","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.10869"}},"official":{"repos":["micahcarroll/unimask"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamical-linear-bandits","slug":"dynamical-linear-bandits","title":"Dynamical Linear Bandits","date":"2022-11-16","arxiv_id":"2211.08997","repositories_listed":1,"syntology":null},{"url":"/paper/agent-state-construction-with-auxiliary","slug":"agent-state-construction-with-auxiliary","title":"Agent-State Construction with Auxiliary Inputs","date":"2022-11-15","arxiv_id":"2211.07805","repositories_listed":1,"syntology":null},{"url":"/paper/doubly-inhomogeneous-reinforcement-learning","slug":"doubly-inhomogeneous-reinforcement-learning","title":"Doubly Inhomogeneous Reinforcement Learning","date":"2022-11-08","arxiv_id":"2211.03983","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-follow-instructions-in-text-based","slug":"learning-to-follow-instructions-in-text-based","title":"Learning to Follow Instructions in Text-Based Games","date":"2022-11-08","arxiv_id":"2211.04591","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-follow-instructions-in-text-based#ran","syntology_url":"https://syntology.ai/paper/2211.04591","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.04591"}},"official":{"repos":["mathieutuli/ltl-gata"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dungeons-and-data-a-large-scale-nethack","slug":"dungeons-and-data-a-large-scale-nethack","title":"Dungeons and Data: A Large-Scale NetHack Dataset","date":"2022-11-01","arxiv_id":"2211.00539","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dungeons-and-data-a-large-scale-nethack#ran","syntology_url":"https://syntology.ai/paper/2211.00539","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.00539"}},"official":{"repos":["facebookresearch/nle"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/popart-efficient-sparse-regression-and","slug":"popart-efficient-sparse-regression-and","title":"PopArt: Efficient Sparse Regression and Experimental Design for Optimal Sparse Linear Bandits","date":"2022-10-25","arxiv_id":"2210.15345","repositories_listed":1,"syntology":null},{"url":"/paper/markup-to-image-diffusion-models-with","slug":"markup-to-image-diffusion-models-with","title":"Markup-to-Image Diffusion Models with Scheduled Sampling","date":"2022-10-11","arxiv_id":"2210.05147","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/markup-to-image-diffusion-models-with#ran","syntology_url":"https://syntology.ai/paper/2210.05147","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05147"}},"official":{"repos":["da03/markup2im"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/continuous-monte-carlo-graph-search","slug":"continuous-monte-carlo-graph-search","title":"Continuous Monte Carlo Graph Search","date":"2022-10-04","arxiv_id":"2210.01426","repositories_listed":1,"syntology":null},{"url":"/paper/non-monotonic-resource-utilization-in-the","slug":"non-monotonic-resource-utilization-in-the","title":"Non-monotonic Resource Utilization in the Bandits with Knapsacks Problem","date":"2022-09-24","arxiv_id":"2209.12013","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/non-monotonic-resource-utilization-in-the#ran","syntology_url":"https://syntology.ai/paper/2209.12013","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.12013"}},"official":{"repos":["raunakkmr/non-monotonic-resource-utilization-in-the-bandits-with-knapsacks-problem-code"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/batch-bayesian-optimisation-via-density-ratio","slug":"batch-bayesian-optimisation-via-density-ratio","title":"Batch Bayesian optimisation via density-ratio estimation with guarantees","date":"2022-09-22","arxiv_id":"2209.10715","repositories_listed":1,"syntology":null},{"url":"/paper/scales-from-fairness-principles-to","slug":"scales-from-fairness-principles-to","title":"SCALES: From Fairness Principles to Constrained Decision-Making","date":"2022-09-22","arxiv_id":"2209.10860","repositories_listed":1,"syntology":null},{"url":"/paper/federated-online-clustering-of-bandits","slug":"federated-online-clustering-of-bandits","title":"Federated Online Clustering of Bandits","date":"2022-08-31","arxiv_id":"2208.14865","repositories_listed":1,"syntology":null},{"url":"/paper/hindsight-learning-for-mdps-with-exogenous","slug":"hindsight-learning-for-mdps-with-exogenous","title":"Hindsight Learning for MDPs with Exogenous Inputs","date":"2022-07-13","arxiv_id":"2207.06272","repositories_listed":1,"syntology":null},{"url":"/paper/contextual-bandits-with-large-action-spaces","slug":"contextual-bandits-with-large-action-spaces","title":"Contextual Bandits with Large Action Spaces: Made Practical","date":"2022-07-12","arxiv_id":"2207.05836","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/contextual-bandits-with-large-action-spaces#ran","syntology_url":"https://syntology.ai/paper/2207.05836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05836"}},"official":{"repos":["pmineiro/linrepcb"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/transformer-neural-processes-uncertainty","slug":"transformer-neural-processes-uncertainty","title":"Transformer Neural Processes: Uncertainty-Aware Meta Learning Via Sequence Modeling","date":"2022-07-09","arxiv_id":"2207.04179","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/transformer-neural-processes-uncertainty#ran","syntology_url":"https://syntology.ai/paper/2207.04179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.04179"}},"official":{"repos":["tung-nd/tnp-pytorch"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/interactively-learning-preference-constraints","slug":"interactively-learning-preference-constraints","title":"Interactively Learning Preference Constraints in Linear Bandits","date":"2022-06-10","arxiv_id":"2206.05255","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/interactively-learning-preference-constraints#ran","syntology_url":"https://syntology.ai/paper/2206.05255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05255"}},"official":{"repos":["lasgroup/adaptive-constraint-learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-anytime-learning-of-markov-decision","slug":"robust-anytime-learning-of-markov-decision","title":"Robust Anytime Learning of Markov Decision Processes","date":"2022-05-31","arxiv_id":"2205.15827","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-is-a","slug":"multi-agent-reinforcement-learning-is-a","title":"Multi-Agent Reinforcement Learning is a Sequence Modeling Problem","date":"2022-05-30","arxiv_id":"2205.14953","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-deviation-types-and-learning-for-1","slug":"efficient-deviation-types-and-learning-for-1","title":"Efficient Deviation Types and Learning for Hindsight Rationality in Extensive-Form Games: Corrections","date":"2022-05-24","arxiv_id":"2205.12031","repositories_listed":1,"syntology":null},{"url":"/paper/explainable-knowledge-graph-embedding","slug":"explainable-knowledge-graph-embedding","title":"Explainable Knowledge Graph Embedding: Inference Reconciliation for Knowledge Inferences Supporting Robot Actions","date":"2022-05-04","arxiv_id":"2205.01836","repositories_listed":1,"syntology":null},{"url":"/paper/evolutionary-multi-armed-bandits-with-genetic","slug":"evolutionary-multi-armed-bandits-with-genetic","title":"Evolutionary Multi-Armed Bandits with Genetic Thompson Sampling","date":"2022-04-26","arxiv_id":"2205.10113","repositories_listed":1,"syntology":null},{"url":"/paper/toward-policy-explanations-for-multi-agent","slug":"toward-policy-explanations-for-multi-agent","title":"Toward Policy Explanations for Multi-Agent Reinforcement Learning","date":"2022-04-26","arxiv_id":"2204.12568","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-a-two-echelon","slug":"deep-reinforcement-learning-for-a-two-echelon","title":"Comparing Deep Reinforcement Learning Algorithms in Two-Echelon Supply Chains","date":"2022-04-20","arxiv_id":"2204.09603","repositories_listed":1,"syntology":null},{"url":"/paper/achieving-long-term-fairness-in-sequential","slug":"achieving-long-term-fairness-in-sequential","title":"Achieving Long-Term Fairness in Sequential Decision Making","date":"2022-04-04","arxiv_id":"2204.01819","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":1,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 3 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/achieving-long-term-fairness-in-sequential#ran","syntology_url":"https://syntology.ai/paper/2204.01819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.01819"}},"official":{"repos":["yaoweihu/achieving-long-term-fairness"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-sandbox-environment-for-generalizable","slug":"the-sandbox-environment-for-generalizable","title":"The Sandbox Environment for Generalizable Agent Research (SEGAR)","date":"2022-03-19","arxiv_id":"2203.10351","repositories_listed":1,"syntology":null},{"url":"/paper/curriculum-based-reinforcement-learning-for","slug":"curriculum-based-reinforcement-learning-for","title":"Curriculum-based Reinforcement Learning for Distribution System Critical Load Restoration","date":"2022-03-08","arxiv_id":"2203.04166","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-entity-1","slug":"deep-reinforcement-learning-for-entity-1","title":"Deep Reinforcement Learning for Entity Alignment","date":"2022-03-07","arxiv_id":"2203.03315","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-for-entity-1#ran","syntology_url":"https://syntology.ai/paper/2203.03315","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.03315"}},"official":{"repos":["guolingbing/rlea"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ai-planning-annotation-for-sample-efficient","slug":"ai-planning-annotation-for-sample-efficient","title":"Hierarchical Reinforcement Learning with AI Planning Models","date":"2022-03-01","arxiv_id":"2203.00669","repositories_listed":1,"syntology":null},{"url":"/paper/lisa-learning-interpretable-skill","slug":"lisa-learning-interpretable-skill","title":"LISA: Learning Interpretable Skill Abstractions from Language","date":"2022-02-28","arxiv_id":"2203.00054","repositories_listed":1,"syntology":null},{"url":"/paper/pre-trained-language-models-for-interactive","slug":"pre-trained-language-models-for-interactive","title":"Pre-Trained Language Models for Interactive Decision-Making","date":"2022-02-03","arxiv_id":"2202.01771","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/pre-trained-language-models-for-interactive#ran","syntology_url":"https://syntology.ai/paper/2202.01771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.01771"}},"official":null}},{"url":"/paper/bellman-meets-hawkes-model-based","slug":"bellman-meets-hawkes-model-based","title":"Bellman Meets Hawkes: Model-Based Reinforcement Learning via Temporal Point Processes","date":"2022-01-29","arxiv_id":"2201.12569","repositories_listed":1,"syntology":null},{"url":"/paper/accelerate-model-parallel-training-by-using","slug":"accelerate-model-parallel-training-by-using","title":"Accelerate Model Parallel Training by Using Efficient Graph Traversal Order in Device Placement","date":"2022-01-21","arxiv_id":"2201.09676","repositories_listed":1,"syntology":null},{"url":"/paper/differentially-private-regret-minimization-in","slug":"differentially-private-regret-minimization-in","title":"Differentially Private Regret Minimization in Episodic Markov Decision Processes","date":"2021-12-20","arxiv_id":"2112.10599","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/differentially-private-regret-minimization-in#ran","syntology_url":"https://syntology.ai/paper/2112.10599","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.10599"}},"official":{"repos":["xingyuzhou989/privatetabularrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autogmap-learning-to-map-large-scale-sparse","slug":"autogmap-learning-to-map-large-scale-sparse","title":"AutoGMap: Learning to Map Large-scale Sparse Graphs on Memristive Crossbars","date":"2021-11-15","arxiv_id":"2111.07684","repositories_listed":1,"syntology":null},{"url":"/paper/sope-spectrum-of-off-policy-estimators","slug":"sope-spectrum-of-off-policy-estimators","title":"SOPE: Spectrum of Off-Policy Estimators","date":"2021-11-06","arxiv_id":"2111.03936","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sope-spectrum-of-off-policy-estimators#ran","syntology_url":"https://syntology.ai/paper/2111.03936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.03936"}},"official":{"repos":["pearl-utexas/sope"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rlds-an-ecosystem-to-generate-share-and-use","slug":"rlds-an-ecosystem-to-generate-share-and-use","title":"RLDS: an Ecosystem to Generate, Share and Use Datasets in Reinforcement Learning","date":"2021-11-04","arxiv_id":"2111.02767","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rlds-an-ecosystem-to-generate-share-and-use#ran","syntology_url":"https://syntology.ai/paper/2111.02767","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.02767"}},"official":{"repos":["google-research/rlds"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/object-aware-regularization-for-addressing","slug":"object-aware-regularization-for-addressing","title":"Object-Aware Regularization for Addressing Causal Confusion in Imitation Learning","date":"2021-10-27","arxiv_id":"2110.14118","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/object-aware-regularization-for-addressing#ran","syntology_url":"https://syntology.ai/paper/2110.14118","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.14118"}},"official":{"repos":["alinlab/oreo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-causal-bayesian-optimization","slug":"dynamic-causal-bayesian-optimization","title":"Dynamic Causal Bayesian Optimization","date":"2021-10-26","arxiv_id":"2110.13891","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dynamic-causal-bayesian-optimization#ran","syntology_url":"https://syntology.ai/paper/2110.13891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.13891"}},"official":{"repos":["neildhir/dcbo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/relace-reinforcement-learning-agent-for","slug":"relace-reinforcement-learning-agent-for","title":"ReLAX: Reinforcement Learning Agent eXplainer for Arbitrary Predictive Models","date":"2021-10-22","arxiv_id":"2110.11960","repositories_listed":1,"syntology":null},{"url":"/paper/show-me-the-whole-world-towards-entire-item","slug":"show-me-the-whole-world-towards-entire-item","title":"Show Me the Whole World: Towards Entire Item Space Exploration for Interactive Personalized Recommendations","date":"2021-10-19","arxiv_id":"2110.09905","repositories_listed":1,"syntology":null},{"url":"/paper/medical-dead-ends-and-learning-to-identify","slug":"medical-dead-ends-and-learning-to-identify","title":"Medical Dead-ends and Learning to Identify High-risk States and Treatments","date":"2021-10-08","arxiv_id":"2110.04186","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-aware-deep-reinforcement-learning","slug":"temporal-aware-deep-reinforcement-learning","title":"Temporal Shift Reinforcement Learning","date":"2021-09-05","arxiv_id":"2109.02145","repositories_listed":1,"syntology":null},{"url":"/paper/rltutor-reinforcement-learning-based-adaptive","slug":"rltutor-reinforcement-learning-based-adaptive","title":"RLTutor: Reinforcement Learning Based Adaptive Tutoring System by Modeling Virtual Student with Fewer Interactions","date":"2021-07-31","arxiv_id":"2108.00268","repositories_listed":1,"syntology":null},{"url":"/paper/neural-contextual-bandits-without-regret","slug":"neural-contextual-bandits-without-regret","title":"Neural Contextual Bandits without Regret","date":"2021-07-07","arxiv_id":"2107.03144","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/neural-contextual-bandits-without-regret#ran","syntology_url":"https://syntology.ai/paper/2107.03144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.03144"}},"official":{"repos":["pkassraie/NNUCB"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/counterfactual-explanations-in-sequential","slug":"counterfactual-explanations-in-sequential","title":"Counterfactual Explanations in Sequential Decision Making Under Uncertainty","date":"2021-07-06","arxiv_id":"2107.02776","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/counterfactual-explanations-in-sequential#ran","syntology_url":"https://syntology.ai/paper/2107.02776","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.02776"}},"official":{"repos":["networks-learning/counterfactual-explanations-mdp"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/bounded-rationality-for-relaxing-best","slug":"bounded-rationality-for-relaxing-best","title":"Bounded rationality for relaxing best response and mutual consistency: The Quantal Hierarchy model of decision-making","date":"2021-06-30","arxiv_id":"2106.15844","repositories_listed":1,"syntology":null},{"url":"/paper/robust-reinforcement-learning-under-minimax","slug":"robust-reinforcement-learning-under-minimax","title":"Robust Reinforcement Learning Under Minimax Regret for Green Security","date":"2021-06-15","arxiv_id":"2106.08413","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-reinforcement-learning-under-minimax#ran","syntology_url":"https://syntology.ai/paper/2106.08413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.08413"}},"official":{"repos":["lily-x/mirror"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cooperative-online-learning","slug":"cooperative-online-learning","title":"Cooperative Online Learning with Feedback Graphs","date":"2021-06-09","arxiv_id":"2106.04982","repositories_listed":1,"syntology":null},{"url":"/paper/curriculum-design-for-teaching-via","slug":"curriculum-design-for-teaching-via","title":"Curriculum Design for Teaching via Demonstrations: Theory and Applications","date":"2021-06-08","arxiv_id":"2106.04696","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":1,"n_ran_checked":2,"n_instrument":6,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":9,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/curriculum-design-for-teaching-via#ran","syntology_url":"https://syntology.ai/paper/2106.04696","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04696"}},"official":{"repos":["adishs/neurips2021_curriculum-teaching-demonstrations_code"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/the-medkit-learn-ing-environment-medical","slug":"the-medkit-learn-ing-environment-medical","title":"The Medkit-Learn(ing) Environment: Medical Decision Modelling through Simulation","date":"2021-06-08","arxiv_id":"2106.04240","repositories_listed":1,"syntology":null},{"url":"/paper/universal-off-policy-evaluation","slug":"universal-off-policy-evaluation","title":"Universal Off-Policy Evaluation","date":"2021-04-26","arxiv_id":"2104.12820","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/universal-off-policy-evaluation#ran","syntology_url":"https://syntology.ai/paper/2104.12820","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.12820"}},"official":{"repos":["yashchandak/UnO"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/independent-reinforcement-learning-for-weakly","slug":"independent-reinforcement-learning-for-weakly","title":"Independent Reinforcement Learning for Weakly Cooperative Multiagent Traffic Control Problem","date":"2021-04-22","arxiv_id":"2104.10917","repositories_listed":1,"syntology":null},{"url":"/paper/ecole-a-library-for-learning-inside-milp","slug":"ecole-a-library-for-learning-inside-milp","title":"Ecole: A Library for Learning Inside MILP Solvers","date":"2021-04-06","arxiv_id":"2104.02828","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-environment-generation-for-1","slug":"adversarial-environment-generation-for-1","title":"Adversarial Environment Generation for Learning to Navigate the Web","date":"2021-03-02","arxiv_id":"2103.01991","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-deviation-types-and-learning-for","slug":"efficient-deviation-types-and-learning-for","title":"Efficient Deviation Types and Learning for Hindsight Rationality in Extensive-Form Games","date":"2021-02-13","arxiv_id":"2102.06973","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-evaluation-of-active-inference","slug":"an-empirical-evaluation-of-active-inference","title":"An empirical evaluation of active inference in multi-armed bandits","date":"2021-01-21","arxiv_id":"2101.08699","repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-optimization-of-portfolio","slug":"off-policy-optimization-of-portfolio","title":"Off-Policy Optimization of Portfolio Allocation Policies under Constraints","date":"2020-12-21","arxiv_id":"2012.11715","repositories_listed":1,"syntology":null},{"url":"/paper/hindsight-and-sequential-rationality-of","slug":"hindsight-and-sequential-rationality-of","title":"Hindsight and Sequential Rationality of Correlated Play","date":"2020-12-10","arxiv_id":"2012.05874","repositories_listed":1,"syntology":null},{"url":"/paper/timeshap-explaining-recurrent-models-through","slug":"timeshap-explaining-recurrent-models-through","title":"TimeSHAP: Explaining Recurrent Models through Sequence Perturbations","date":"2020-11-30","arxiv_id":"2012.00073","repositories_listed":1,"syntology":null},{"url":"/paper/lava-latent-action-spaces-via-variational","slug":"lava-latent-action-spaces-via-variational","title":"LAVA: Latent Action Spaces via Variational Auto-encoding for Dialogue Policy Optimization","date":"2020-11-18","arxiv_id":"2011.09378","repositories_listed":1,"syntology":null},{"url":"/paper/corrupted-contextual-bandits-with-action","slug":"corrupted-contextual-bandits-with-action","title":"A New Bandit Setting Balancing Information from State Evolution and Corrupted Context","date":"2020-11-16","arxiv_id":"2011.07989","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-stress-testing-of-trajectory","slug":"adaptive-stress-testing-of-trajectory","title":"Adaptive Stress Testing of Trajectory Predictions in Flight Management Systems","date":"2020-11-04","arxiv_id":"2011.02559","repositories_listed":1,"syntology":null},{"url":"/paper/loss-bounds-for-approximate-influence-based","slug":"loss-bounds-for-approximate-influence-based","title":"Loss Bounds for Approximate Influence-Based Abstraction","date":"2020-11-03","arxiv_id":"2011.01788","repositories_listed":1,"syntology":null},{"url":"/paper/towards-safe-policy-improvement-for-non","slug":"towards-safe-policy-improvement-for-non","title":"Towards Safe Policy Improvement for Non-Stationary MDPs","date":"2020-10-23","arxiv_id":"2010.12645","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-generalize-for-sequential","slug":"learning-to-generalize-for-sequential","title":"Learning to Generalize for Sequential Decision Making","date":"2020-10-05","arxiv_id":"2010.02229","repositories_listed":1,"syntology":null},{"url":"/paper/multi-task-causal-learning-with-gaussian","slug":"multi-task-causal-learning-with-gaussian","title":"Multi-task Causal Learning with Gaussian Processes","date":"2020-09-27","arxiv_id":"2009.12821","repositories_listed":1,"syntology":null},{"url":"/paper/certrl-formalizing-convergence-proofs-for","slug":"certrl-formalizing-convergence-proofs-for","title":"CertRL: Formalizing Convergence Proofs for Value and Policy Iteration in Coq","date":"2020-09-23","arxiv_id":"2009.11403","repositories_listed":1,"syntology":null},{"url":"/paper/occupancy-anticipation-for-efficient","slug":"occupancy-anticipation-for-efficient","title":"Occupancy Anticipation for Efficient Exploration and Navigation","date":"2020-08-21","arxiv_id":"2008.09285","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/occupancy-anticipation-for-efficient#ran","syntology_url":"https://syntology.ai/paper/2008.09285","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.09285"}},"official":{"repos":["facebookresearch/OccupancyAnticipation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fast-reinforcement-learning-with-generalized","slug":"fast-reinforcement-learning-with-generalized","title":"Fast reinforcement learning with generalized policy updates","date":"2020-07-09","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/enforcing-almost-sure-reachability-in-pomdps","slug":"enforcing-almost-sure-reachability-in-pomdps","title":"Enforcing Almost-Sure Reachability in POMDPs","date":"2020-06-30","arxiv_id":"2007.00085","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-nonmyopic-bayesian-optimization-via","slug":"efficient-nonmyopic-bayesian-optimization-via","title":"Efficient Nonmyopic Bayesian Optimization via One-Shot Multi-Step Trees","date":"2020-06-29","arxiv_id":"2006.15779","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/efficient-nonmyopic-bayesian-optimization-via#ran","syntology_url":"https://syntology.ai/paper/2006.15779","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.15779"}},"official":null}},{"url":"/paper/frequentist-uncertainty-in-recurrent-neural","slug":"frequentist-uncertainty-in-recurrent-neural","title":"Frequentist Uncertainty in Recurrent Neural Networks via Blockwise Influence Functions","date":"2020-06-20","arxiv_id":"2006.13707","repositories_listed":1,"syntology":null},{"url":"/paper/mutual-information-based-knowledge-transfer","slug":"mutual-information-based-knowledge-transfer","title":"Mutual Information Based Knowledge Transfer Under State-Action Dimension Mismatch","date":"2020-06-12","arxiv_id":"2006.07041","repositories_listed":1,"syntology":null},{"url":"/paper/recurrent-sum-product-max-networks-for","slug":"recurrent-sum-product-max-networks-for","title":"Recurrent Sum-Product-Max Networks for Decision Making in Perfectly-Observed Environments","date":"2020-06-12","arxiv_id":"2006.07300","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning","slug":"reinforcement-learning","title":"Reinforcement Learning","date":"2020-05-29","arxiv_id":"2005.14419","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-multi-robot-task-allocation-under","slug":"dynamic-multi-robot-task-allocation-under","title":"Dynamic Multi-Robot Task Allocation under Uncertainty and Temporal Constraints","date":"2020-05-27","arxiv_id":"2005.13109","repositories_listed":1,"syntology":null},{"url":"/paper/think-too-fast-nor-too-slow-the-computational","slug":"think-too-fast-nor-too-slow-the-computational","title":"Think Too Fast Nor Too Slow: The Computational Trade-off Between Planning And Reinforcement Learning","date":"2020-05-15","arxiv_id":"2005.07404","repositories_listed":1,"syntology":null},{"url":"/paper/unified-models-of-human-behavioral-agents-in","slug":"unified-models-of-human-behavioral-agents-in","title":"Unified Models of Human Behavioral Agents in Bandits, Contextual Bandits and RL","date":"2020-05-10","arxiv_id":"2005.04544","repositories_listed":1,"syntology":null},{"url":"/paper/learning-sparse-rewarded-tasks-from-sub","slug":"learning-sparse-rewarded-tasks-from-sub","title":"Learning Sparse Rewarded Tasks from Sub-Optimal Demonstrations","date":"2020-04-01","arxiv_id":"2004.00530","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-sparse-rewarded-tasks-from-sub#ran","syntology_url":"https://syntology.ai/paper/2004.00530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.00530"}},"official":null}},{"url":"/paper/off-policy-policy-evaluation-for-sequential","slug":"off-policy-policy-evaluation-for-sequential","title":"Off-policy Policy Evaluation For Sequential Decisions Under Unobserved Confounding","date":"2020-03-12","arxiv_id":"2003.05623","repositories_listed":1,"syntology":null},{"url":"/paper/learning-discrete-state-abstractions-with","slug":"learning-discrete-state-abstractions-with","title":"Learning Discrete State Abstractions With Deep Variational Inference","date":"2020-03-09","arxiv_id":"2003.04300","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-discrete-state-abstractions-with#ran","syntology_url":"https://syntology.ai/paper/2003.04300","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.04300"}},"official":{"repos":["ondrejba/discrete_abstractions"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/can-increasing-input-dimensionality-improve","slug":"can-increasing-input-dimensionality-improve","title":"Can Increasing Input Dimensionality Improve Deep Reinforcement Learning?","date":"2020-03-03","arxiv_id":"2003.01629","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-of-risk-constrained","slug":"reinforcement-learning-of-risk-constrained","title":"Reinforcement Learning of Risk-Constrained Policies in Markov Decision Processes","date":"2020-02-27","arxiv_id":"2002.12086","repositories_listed":1,"syntology":null},{"url":"/paper/learning-dynamic-knowledge-graphs-to","slug":"learning-dynamic-knowledge-graphs-to","title":"Learning Dynamic Belief Graphs to Generalize on Text-Based Games","date":"2020-02-21","arxiv_id":"2002.09127","repositories_listed":1,"syntology":null}],"record_sha256":"2d3de2c330de3935f07490e7fc050e6bf3a9687ccc64d3edfb92919c6115dc38","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}