{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/sequential-decision-making/papers/4","list_of":"/task/sequential-decision-making","task":"Sequential Decision Making","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":13,"rows_per_page":100,"rows":[301,400],"of":1210,"counts":{"archive_papers_tagged":1210,"with_a_code_link":351,"where_syntology_ran_a_sample":107,"not_listed_spam_title":0,"listed":1210,"listed_where_code_ran":107,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":90,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":90,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/sequential-decision-making","prev":"/task/sequential-decision-making/papers/3","next":"/task/sequential-decision-making/papers/5","papers":[{"url":"/paper/effective-reinforcement-learning-through","slug":"effective-reinforcement-learning-through","title":"Effective Reinforcement Learning through Evolutionary Surrogate-Assisted Prescription","date":"2020-02-13","arxiv_id":"2002.05368","repositories_listed":1,"syntology":null},{"url":"/paper/does-the-markov-decision-process-fit-the-data","slug":"does-the-markov-decision-process-fit-the-data","title":"Does the Markov Decision Process Fit the Data: Testing for the Markov Property in Sequential Decision Making","date":"2020-02-05","arxiv_id":"2002.01751","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 4 unverified","sample_list":"/paper/does-the-markov-decision-process-fit-the-data#ran","syntology_url":"https://syntology.ai/paper/2002.01751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.01751"}},"official":null}},{"url":"/paper/computing-the-feedback-capacity-of-finite","slug":"computing-the-feedback-capacity-of-finite","title":"Computing the Feedback Capacity of Finite State Channels using Reinforcement Learning","date":"2020-01-27","arxiv_id":"2001.09685","repositories_listed":1,"syntology":null},{"url":"/paper/statistical-inference-of-the-value-function","slug":"statistical-inference-of-the-value-function","title":"Statistical Inference of the Value Function for Reinforcement Learning in Infinite Horizon Settings","date":"2020-01-13","arxiv_id":"2001.04515","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/statistical-inference-of-the-value-function#ran","syntology_url":"https://syntology.ai/paper/2001.04515","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.04515"}},"official":{"repos":["shengzhang37/SAVE"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/risk-averse-action-selection-using-extreme","slug":"risk-averse-action-selection-using-extreme","title":"Risk-Averse Action Selection Using Extreme Value Theory Estimates of the CVaR","date":"2019-12-03","arxiv_id":"1912.01718","repositories_listed":1,"syntology":null},{"url":"/paper/smile-scalable-meta-inverse-reinforcement","slug":"smile-scalable-meta-inverse-reinforcement","title":"SMILe: Scalable Meta Inverse Reinforcement Learning through Context-Conditional Policies","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/planning-with-goal-conditioned-policies-1","slug":"planning-with-goal-conditioned-policies-1","title":"Planning with Goal-Conditioned Policies","date":"2019-11-19","arxiv_id":"1911.08453","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/planning-with-goal-conditioned-policies-1#ran","syntology_url":"https://syntology.ai/paper/1911.08453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.08453"}},"official":null}},{"url":"/paper/a-biologically-plausible-benchmark-for","slug":"a-biologically-plausible-benchmark-for","title":"A Biologically Plausible Benchmark for Contextual Bandit Algorithms in Precision Oncology Using in vitro Data","date":"2019-11-11","arxiv_id":"1911.04389","repositories_listed":1,"syntology":null},{"url":"/paper/thompson-sampling-for-contextual-bandit","slug":"thompson-sampling-for-contextual-bandit","title":"Thompson Sampling for Contextual Bandit Problems with Auxiliary Safety Constraints","date":"2019-11-02","arxiv_id":"1911.00638","repositories_listed":1,"syntology":null},{"url":"/paper/thompson-sampling-via-local-uncertainty","slug":"thompson-sampling-via-local-uncertainty","title":"Thompson Sampling via Local Uncertainty","date":"2019-10-30","arxiv_id":"1910.13673","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/thompson-sampling-via-local-uncertainty#ran","syntology_url":"https://syntology.ai/paper/1910.13673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.13673"}},"official":{"repos":["Zhendong-Wang/Thompson-Sampling-via-Local-Uncertainty"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/approximate-inference-in-discrete","slug":"approximate-inference-in-discrete","title":"Approximate Inference in Discrete Distributions with Monte Carlo Tree Search and Value Functions","date":"2019-10-15","arxiv_id":"1910.06862","repositories_listed":1,"syntology":null},{"url":"/paper/deep-q-network-for-angry-birds","slug":"deep-q-network-for-angry-birds","title":"Deep Q-Network for Angry Birds","date":"2019-10-04","arxiv_id":"1910.01806","repositories_listed":1,"syntology":null},{"url":"/paper/mabwiser-a-parallelizable-contextual-multi","slug":"mabwiser-a-parallelizable-contextual-multi","title":"MABWiser: A Parallelizable Contextual Multi-Armed Bandit Library for Python","date":"2019-10-04","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-temporal-logic","slug":"reinforcement-learning-for-temporal-logic","title":"Reinforcement Learning for Temporal Logic Control Synthesis with Probabilistic Satisfaction Guarantees","date":"2019-09-11","arxiv_id":"1909.05304","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-for-temporal-logic#ran","syntology_url":"https://syntology.ai/paper/1909.05304","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.05304"}},"official":{"repos":["grockious/lcrl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/back-to-the-future-sequential-alignment-of","slug":"back-to-the-future-sequential-alignment-of","title":"Back to the Future -- Sequential Alignment of Text Representations","date":"2019-09-08","arxiv_id":"1909.03464","repositories_listed":1,"syntology":null},{"url":"/paper/prediction-consistency-curvature","slug":"prediction-consistency-curvature","title":"Prediction, Consistency, Curvature: Representation Learning for Locally-Linear Control","date":"2019-09-04","arxiv_id":"1909.01506","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/prediction-consistency-curvature#ran","syntology_url":"https://syntology.ai/paper/1909.01506","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.01506"}},"official":null}},{"url":"/paper/interactive-machine-comprehension-with","slug":"interactive-machine-comprehension-with","title":"Interactive Machine Comprehension with Information Seeking Agents","date":"2019-08-27","arxiv_id":"1908.10449","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/interactive-machine-comprehension-with#ran","syntology_url":"https://syntology.ai/paper/1908.10449","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.10449"}},"official":{"repos":["xingdi-eric-yuan/imrc_public"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-learning-for-efficient-reinforcement","slug":"reward-learning-for-efficient-reinforcement","title":"Reward Learning for Efficient Reinforcement Learning in Extractive Document Summarisation","date":"2019-07-30","arxiv_id":"1907.12894","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reward-learning-for-efficient-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1907.12894","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.12894"}},"official":{"repos":["UKPLab/ijcai2019-relis"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-multi-armed-bandit-algorithms","slug":"scaling-multi-armed-bandit-algorithms","title":"Scaling Multi-Armed Bandit Algorithms","date":"2019-07-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/co-training-for-policy-learning","slug":"co-training-for-policy-learning","title":"Co-training for Policy Learning","date":"2019-07-03","arxiv_id":"1907.04484","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-by-word-image-grounded-vocabulary","slug":"bridging-by-word-image-grounded-vocabulary","title":"Bridging by Word: Image Grounded Vocabulary Construction for Visual Captioning","date":"2019-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-hierarchical-architecture-for-sequential","slug":"a-hierarchical-architecture-for-sequential","title":"A Hierarchical Architecture for Sequential Decision-Making in Autonomous Driving using Deep Reinforcement Learning","date":"2019-06-20","arxiv_id":"1906.08464","repositories_listed":1,"syntology":null},{"url":"/paper/lifelong-learning-with-a-changing-action-set","slug":"lifelong-learning-with-a-changing-action-set","title":"Lifelong Learning with a Changing Action Set","date":"2019-06-05","arxiv_id":"1906.01770","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-when-all-actions-are","slug":"reinforcement-learning-when-all-actions-are","title":"Reinforcement Learning When All Actions are Not Always Available","date":"2019-06-05","arxiv_id":"1906.01772","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-discretize-solving-1d-scalar","slug":"learning-to-discretize-solving-1d-scalar","title":"Learning to Discretize: Solving 1D Scalar Conservation Laws via Deep Reinforcement Learning","date":"2019-05-27","arxiv_id":"1905.11079","repositories_listed":1,"syntology":null},{"url":"/paper/multi-hop-reading-comprehension-via-deep","slug":"multi-hop-reading-comprehension-via-deep","title":"Multi-hop Reading Comprehension via Deep Reinforcement Learning based Document Traversal","date":"2019-05-23","arxiv_id":"1905.09438","repositories_listed":1,"syntology":null},{"url":"/paper/the-minerl-competition-on-sample-efficient","slug":"the-minerl-competition-on-sample-efficient","title":"The MineRL 2019 Competition on Sample Efficient Reinforcement Learning using Human Priors","date":"2019-04-22","arxiv_id":"1904.10079","repositories_listed":1,"syntology":null},{"url":"/paper/quizbowl-the-case-for-incremental-question","slug":"quizbowl-the-case-for-incremental-question","title":"Quizbowl: The Case for Incremental Question Answering","date":"2019-04-09","arxiv_id":"1904.04792","repositories_listed":1,"syntology":{"n":14,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quizbowl-the-case-for-incremental-question#ran","syntology_url":"https://syntology.ai/paper/1904.04792","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.04792"}},"official":null}},{"url":"/paper/adaptive-sequence-submodularity","slug":"adaptive-sequence-submodularity","title":"Adaptive Sequence Submodularity","date":"2019-02-15","arxiv_id":"1902.05981","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-real-time-multimodal-routing-with","slug":"dynamic-real-time-multimodal-routing-with","title":"Dynamic Real-time Multimodal Routing with Hierarchical Hybrid Planning","date":"2019-02-05","arxiv_id":"1902.01560","repositories_listed":1,"syntology":null},{"url":"/paper/fairness-with-dynamics","slug":"fairness-with-dynamics","title":"Algorithms for Fairness in Sequential Decision Making","date":"2019-01-24","arxiv_id":"1901.08568","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fairness-with-dynamics#ran","syntology_url":"https://syntology.ai/paper/1901.08568","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08568"}},"official":{"repos":["wmgithub/fairness"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/read-watch-and-move-reinforcement-learning","slug":"read-watch-and-move-reinforcement-learning","title":"Read, Watch, and Move: Reinforcement Learning for Temporally Grounding Natural Language Descriptions in Videos","date":"2019-01-21","arxiv_id":"1901.06829","repositories_listed":1,"syntology":null},{"url":"/paper/structural-causal-bandits-where-to-intervene","slug":"structural-causal-bandits-where-to-intervene","title":"Structural Causal Bandits: Where to Intervene?","date":"2018-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/heteroscedastic-bandits-with-reneging","slug":"heteroscedastic-bandits-with-reneging","title":"Stay With Me: Lifetime Maximization Through Heteroscedastic Linear Bandits With Reneging","date":"2018-10-29","arxiv_id":"1810.12418","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-sequence-labeling-with-actor-critic","slug":"efficient-sequence-labeling-with-actor-critic","title":"Efficient Sequence Labeling with Actor-Critic Training","date":"2018-09-30","arxiv_id":"1810.00428","repositories_listed":1,"syntology":null},{"url":"/paper/sequential-importance-sampling-bandits","slug":"sequential-importance-sampling-bandits","title":"Sequential Monte Carlo Bandits","date":"2018-08-08","arxiv_id":"1808.02933","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-listen-read-and-follow-score","slug":"learning-to-listen-read-and-follow-score","title":"Learning to Listen, Read, and Follow: Score Following as a Reinforcement Learning Game","date":"2018-07-17","arxiv_id":"1807.06391","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/learning-to-listen-read-and-follow-score#ran","syntology_url":"https://syntology.ai/paper/1807.06391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.06391"}},"official":{"repos":["CPJKU/score_following_game"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-surgical","slug":"deep-reinforcement-learning-for-surgical","title":"Deep Reinforcement Learning for Surgical Gesture Segmentation and Classification","date":"2018-06-21","arxiv_id":"1806.08089","repositories_listed":1,"syntology":null},{"url":"/paper/deep-variational-reinforcement-learning-for","slug":"deep-variational-reinforcement-learning-for","title":"Deep Variational Reinforcement Learning for POMDPs","date":"2018-06-06","arxiv_id":"1806.02426","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-variational-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/1806.02426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.02426"}},"official":{"repos":["maximilianigl/DVRL"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-the-accuracy-and-fairness-of-human","slug":"enhancing-the-accuracy-and-fairness-of-human","title":"Enhancing the Accuracy and Fairness of Human Decision Making","date":"2018-05-25","arxiv_id":"1805.10318","repositories_listed":1,"syntology":null},{"url":"/paper/machine-teaching-for-inverse-reinforcement","slug":"machine-teaching-for-inverse-reinforcement","title":"Machine Teaching for Inverse Reinforcement Learning: Algorithms and Applications","date":"2018-05-20","arxiv_id":"1805.07687","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/machine-teaching-for-inverse-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1805.07687","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.07687"}},"official":{"repos":["dsbrown1331/machine-teaching-irl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-learning-intrinsic-rewards-for-policy","slug":"on-learning-intrinsic-rewards-for-policy","title":"On Learning Intrinsic Rewards for Policy Gradient Methods","date":"2018-04-17","arxiv_id":"1804.06459","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/on-learning-intrinsic-rewards-for-policy#ran","syntology_url":"https://syntology.ai/paper/1804.06459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.06459"}},"official":{"repos":["Hwhitetooth/lirpg"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/structured-control-nets-for-deep","slug":"structured-control-nets-for-deep","title":"Structured Control Nets for Deep Reinforcement Learning","date":"2018-02-22","arxiv_id":"1802.08311","repositories_listed":1,"syntology":null},{"url":"/paper/utility-decomposition-with-deep-corrections","slug":"utility-decomposition-with-deep-corrections","title":"Decomposition Methods with Deep Corrections for Reinforcement Learning","date":"2018-02-06","arxiv_id":"1802.01772","repositories_listed":1,"syntology":null},{"url":"/paper/learning-structural-weight-uncertainty-for","slug":"learning-structural-weight-uncertainty-for","title":"Learning Structural Weight Uncertainty for Sequential Decision-Making","date":"2017-12-30","arxiv_id":"1801.00085","repositories_listed":1,"syntology":null},{"url":"/paper/classification-with-costly-features-using","slug":"classification-with-costly-features-using","title":"Classification with Costly Features using Deep Reinforcement Learning","date":"2017-11-20","arxiv_id":"1711.07364","repositories_listed":1,"syntology":null},{"url":"/paper/on-improving-deep-reinforcement-learning-for","slug":"on-improving-deep-reinforcement-learning-for","title":"On Improving Deep Reinforcement Learning for POMDPs","date":"2017-04-26","arxiv_id":"1704.07978","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-goal-driven-web-navigation","slug":"end-to-end-goal-driven-web-navigation","title":"End-to-End Goal-Driven Web Navigation","date":"2016-02-06","arxiv_id":"1602.02261","repositories_listed":1,"syntology":null},{"url":"/paper/data-generation-as-sequential-decision-making","slug":"data-generation-as-sequential-decision-making","title":"Data Generation as Sequential Decision Making","date":"2015-06-10","arxiv_id":"1506.03504","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/data-generation-as-sequential-decision-making#ran","syntology_url":"https://syntology.ai/paper/1506.03504","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1506.03504"}},"official":{"repos":["Philip-Bachman/Sequential-Generation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/doubly-robust-policy-evaluation-and","slug":"doubly-robust-policy-evaluation-and","title":"Doubly Robust Policy Evaluation and Optimization","date":"2015-03-10","arxiv_id":"1503.02834","repositories_listed":1,"syntology":null},{"url":"/paper/monte-carlo-sampling-for-regret-minimization","slug":"monte-carlo-sampling-for-regret-minimization","title":"Monte Carlo Sampling for Regret Minimization in Extensive Games","date":"2009-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":null,"slug":"airllm-diffusion-policy-based-adaptive-lora","title":"AirLLM: Diffusion Policy-based Adaptive LoRA for Remote Fine-Tuning of LLM over the Air","date":"2025-07-15","arxiv_id":"2507.11515","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-stackelberg-games-conjectural-reasoning","title":"LLM-Stackelberg Games: Conjectural Reasoning Equilibria and Their Applications to Spearphishing","date":"2025-07-12","arxiv_id":"2507.09407","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-continual-reinforcement-learning","title":"A Survey of Continual Reinforcement Learning","date":"2025-06-27","arxiv_id":"2506.21872","repositories_listed":0,"syntology":null},{"url":"/paper/flow-based-single-step-completion-for","slug":"flow-based-single-step-completion-for","title":"Flow-Based Single-Step Completion for Efficient and Expressive Policy Learning","date":"2025-06-26","arxiv_id":"2506.21427","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/flow-based-single-step-completion-for#ran","syntology_url":"https://syntology.ai/paper/2506.21427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.21427"}},"official":null}},{"url":null,"slug":"polar-a-pessimistic-model-based-policy","title":"POLAR: A Pessimistic Model-based Policy Learning Algorithm for Dynamic Treatment Regimes","date":"2025-06-25","arxiv_id":"2506.20406","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-strategy-synthesis-for-mdps-via","title":"Efficient Strategy Synthesis for MDPs via Hierarchical Block Decomposition","date":"2025-06-21","arxiv_id":"2506.17792","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-armed-bandits-with-machine-learning","title":"Multi-Armed Bandits With Machine Learning-Generated Surrogate Rewards","date":"2025-06-20","arxiv_id":"2506.16658","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-in-context-learning-for-language","title":"Leveraging In-Context Learning for Language Model Agents","date":"2025-06-16","arxiv_id":"2506.13109","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-clustering-of-neural-bandits","title":"Revisiting Clustering of Neural Bandits: Selective Reinitialization for Mitigating Loss of Plasticity","date":"2025-06-14","arxiv_id":"2506.12389","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-09562","title":"TooBadRL: Trigger Optimization to Boost Effectiveness of Backdoor Attacks on Deep Reinforcement Learning","date":"2025-06-11","arxiv_id":"2506.09562","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-responsible-ai-advances-in-safety","title":"Towards Responsible AI: Advances in Safety, Fairness, and Accountability of Autonomous Systems","date":"2025-06-11","arxiv_id":"2506.10192","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-08463","title":"How to Provably Improve Return Conditioned Supervised Learning?","date":"2025-06-10","arxiv_id":"2506.08463","repositories_listed":0,"syntology":null},{"url":null,"slug":"qforce-rl-quantized-fpga-optimized","title":"QForce-RL: Quantized FPGA-Optimized Reinforcement Learning Compute Engine","date":"2025-06-08","arxiv_id":"2506.07046","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-experience-replay-for-self","title":"Contextual Experience Replay for Self-Improvement of Language Agents","date":"2025-06-07","arxiv_id":"2506.06698","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoqd-automatic-discovery-of-diverse","title":"AutoQD: Automatic Discovery of Diverse Behaviors with Quality-Diversity Optimization","date":"2025-06-05","arxiv_id":"2506.05634","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-layer-contrastive-decoding-reduces","title":"Active Layer-Contrastive Decoding Reduces Hallucination in Large Language Model Generation","date":"2025-05-29","arxiv_id":"2505.23657","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergent-risk-awareness-in-rational-agents","title":"Emergent Risk Awareness in Rational Agents under Resource Constraints","date":"2025-05-29","arxiv_id":"2505.23436","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-frontier-exploration-on-graphs-with","title":"Adaptive Frontier Exploration on Graphs with Applications to Network-Based Disease Testing","date":"2025-05-27","arxiv_id":"2505.21671","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-deep-learning-via-implicit","title":"Variational Deep Learning via Implicit Regularization","date":"2025-05-26","arxiv_id":"2505.20235","repositories_listed":0,"syntology":null},{"url":null,"slug":"ddo-dual-decision-optimization-via-multi","title":"DDO: Dual-Decision Optimization via Multi-Agent Collaboration for LLM-Based Medical Consultation","date":"2025-05-24","arxiv_id":"2505.18630","repositories_listed":0,"syntology":null},{"url":null,"slug":"automata-learning-of-preferences-over","title":"Automata Learning of Preferences over Temporal Logic Formulas from Pairwise Comparisons","date":"2025-05-23","arxiv_id":"2505.18030","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-is-enough-llms-are-in-context","title":"Reward Is Enough: LLMs Are In-Context Reinforcement Learners","date":"2025-05-21","arxiv_id":"2506.06303","repositories_listed":0,"syntology":null},{"url":null,"slug":"vid2world-crafting-video-diffusion-models-to","title":"Vid2World: Crafting Video Diffusion Models to Interactive World Models","date":"2025-05-20","arxiv_id":"2505.14357","repositories_listed":0,"syntology":null},{"url":null,"slug":"omgpt-a-sequence-modeling-framework-for-data","title":"OMGPT: A Sequence Modeling Framework for Data-driven Operational Decision Making","date":"2025-05-19","arxiv_id":"2505.13580","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10762","title":"Deep Symbolic Optimization: Reinforcement Learning for Symbolic Mathematics","date":"2025-05-16","arxiv_id":"2505.10762","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalization-guarantees-for-learning-branch","title":"Generalization Guarantees for Learning Branch-and-Cut Policies in Integer Programming","date":"2025-05-16","arxiv_id":"2505.11636","repositories_listed":0,"syntology":null},{"url":null,"slug":"batched-nonparametric-bandits-via-k-nearest","title":"Batched Nonparametric Bandits via k-Nearest Neighbor UCB","date":"2025-05-15","arxiv_id":"2505.10498","repositories_listed":0,"syntology":null},{"url":null,"slug":"counterfactual-strategies-for-markov-decision","title":"Counterfactual Strategies for Markov Decision Processes","date":"2025-05-14","arxiv_id":"2505.09412","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequential-treatment-effect-estimation-with","title":"Sequential Treatment Effect Estimation with Unmeasured Confounders","date":"2025-05-14","arxiv_id":"2505.09113","repositories_listed":0,"syntology":null},{"url":null,"slug":"textsc-rfpg-robust-finite-memory-policy","title":"\\textsc{rfPG}: Robust Finite-Memory Policy Gradients for Hidden-Model POMDPs","date":"2025-05-14","arxiv_id":"2505.09518","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-practical-introduction-to-deep","title":"A Practical Introduction to Deep Reinforcement Learning","date":"2025-05-13","arxiv_id":"2505.08295","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-reinforcement-learning-agents","title":"Explainable Reinforcement Learning Agents Using World Models","date":"2025-05-12","arxiv_id":"2505.08073","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-agent-reinforcement-learning-approach-2","title":"A Multi-Agent Reinforcement Learning Approach for Cooperative Air-Ground-Human Crowdsensing in Emergency Rescue","date":"2025-05-11","arxiv_id":"2505.06997","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-online-decision-making-with","title":"Constrained Online Decision-Making: A Unified Framework","date":"2025-05-11","arxiv_id":"2505.07101","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-daunce-reinforcement-learning-driven-data","title":"RL-DAUNCE: Reinforcement Learning-Driven Data Assimilation with Uncertainty-Aware Constrained Ensembles","date":"2025-05-08","arxiv_id":"2505.05452","repositories_listed":0,"syntology":null},{"url":null,"slug":"mdps-with-a-state-sensing-cost","title":"MDPs with a State Sensing Cost","date":"2025-05-06","arxiv_id":"2505.03280","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-labeled-preference-learning-is","title":"Policy-labeled Preference Learning: Is Preference Enough for RLHF?","date":"2025-05-06","arxiv_id":"2505.06273","repositories_listed":0,"syntology":null},{"url":null,"slug":"d3hrl-a-distributed-hierarchical","title":"D3HRL: A Distributed Hierarchical Reinforcement Learning Approach Based on Causal Discovery and Spurious Correlation Detection","date":"2025-05-04","arxiv_id":"2505.01979","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-learning-of-the-optimal-action-value","title":"Bayesian learning of the optimal action-value function in a Markov decision process","date":"2025-05-03","arxiv_id":"2505.01859","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-minimax-mdp-framework-with-future-imposed","title":"A Minimax-MDP Framework with Future-imposed Conditions for Learning-augmented Problems","date":"2025-05-02","arxiv_id":"2505.00973","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-generated-in-context-examples-improve","title":"Self-Generated In-Context Examples Improve LLM Agents for Sequential Decision-Making Tasks","date":"2025-05-01","arxiv_id":"2505.00234","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-and-robust-task-sampling-with-posterior","title":"Fast and Robust: Task Sampling with Posterior and Diversity Synergies for Adaptive Decision-Makers in Randomized Environments","date":"2025-04-27","arxiv_id":"2504.19139","repositories_listed":0,"syntology":null},{"url":null,"slug":"sapo-rl-sequential-actuator-placement","title":"SAPO-RL: Sequential Actuator Placement Optimization for Fuselage Assembly via Reinforcement Learning","date":"2025-04-24","arxiv_id":"2504.17603","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-attention-fusion-of-visual-and","title":"Hierarchical Attention Fusion of Visual and Textual Representations for Cross-Domain Sequential Recommendation","date":"2025-04-21","arxiv_id":"2504.15085","repositories_listed":0,"syntology":null},{"url":null,"slug":"consensus-in-motion-a-case-of-dynamic","title":"Consensus in Motion: A Case of Dynamic Rationality of Sequential Learning in Probability Aggregation","date":"2025-04-20","arxiv_id":"2504.14624","repositories_listed":0,"syntology":null},{"url":null,"slug":"tales-text-adventure-learning-environment","title":"TALES: Text Adventure Learning Environment Suite","date":"2025-04-19","arxiv_id":"2504.14128","repositories_listed":0,"syntology":null},{"url":null,"slug":"position-paper-rethinking-privacy-in-rl-for","title":"Position Paper: Rethinking Privacy in RL for Sequential Decision-making in the Age of LLMs","date":"2025-04-15","arxiv_id":"2504.11511","repositories_listed":0,"syntology":null},{"url":null,"slug":"truncated-matrix-completion-an-empirical","title":"Truncated Matrix Completion - An Empirical Study","date":"2025-04-14","arxiv_id":"2504.09873","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-more-efficient-robust-instance","title":"Towards More Efficient, Robust, Instance-adaptive, and Generalizable Sequential Decision making","date":"2025-04-12","arxiv_id":"2504.09192","repositories_listed":0,"syntology":null}],"record_sha256":"41458a3b5a2f14e91f936b7238a3a259649030815d9b09a919825889f044a13c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}