{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/sequential-decision-making/papers/2","list_of":"/task/sequential-decision-making","task":"Sequential Decision Making","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":13,"rows_per_page":100,"rows":[101,200],"of":1210,"counts":{"archive_papers_tagged":1210,"with_a_code_link":351,"where_syntology_ran_a_sample":107,"not_listed_spam_title":0,"listed":1210,"listed_where_code_ran":107,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":90,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":90,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/sequential-decision-making","prev":"/task/sequential-decision-making","next":"/task/sequential-decision-making/papers/3","papers":[{"url":"/paper/combining-experimental-and-historical-data","slug":"combining-experimental-and-historical-data","title":"Combining Experimental and Historical Data for Policy Evaluation","date":"2024-06-01","arxiv_id":"2406.00317","repositories_listed":1,"syntology":null},{"url":"/paper/pursuing-overall-welfare-in-federated","slug":"pursuing-overall-welfare-in-federated","title":"Pursuing Overall Welfare in Federated Learning through Sequential Decision Making","date":"2024-05-31","arxiv_id":"2405.20821","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pursuing-overall-welfare-in-federated#ran","syntology_url":"https://syntology.ai/paper/2405.20821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20821"}},"official":{"repos":["vaseline555/aaggff"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-machines-for-deep-rl-in-noisy-and","slug":"reward-machines-for-deep-rl-in-noisy-and","title":"Reward Machines for Deep RL in Noisy and Uncertain Environments","date":"2024-05-31","arxiv_id":"2406.00120","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/reward-machines-for-deep-rl-in-noisy-and#ran","syntology_url":"https://syntology.ai/paper/2406.00120","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00120"}},"official":{"repos":["andrewli77/reward-machines-noisy-environments"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-transformers-in-solving-pomdps","slug":"rethinking-transformers-in-solving-pomdps","title":"Rethinking Transformers in Solving POMDPs","date":"2024-05-27","arxiv_id":"2405.17358","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/rethinking-transformers-in-solving-pomdps#ran","syntology_url":"https://syntology.ai/paper/2405.17358","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17358"}},"official":{"repos":["ctp314/tfporl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/fliphat-joint-differential-privacy-for-high","slug":"fliphat-joint-differential-privacy-for-high","title":"FLIPHAT: Joint Differential Privacy for High Dimensional Sparse Linear Bandits","date":"2024-05-22","arxiv_id":"2405.14038","repositories_listed":1,"syntology":null},{"url":"/paper/is-mamba-compatible-with-trajectory","slug":"is-mamba-compatible-with-trajectory","title":"Is Mamba Compatible with Trajectory Optimization in Offline Reinforcement Learning?","date":"2024-05-20","arxiv_id":"2405.12094","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/is-mamba-compatible-with-trajectory#ran","syntology_url":"https://syntology.ai/paper/2405.12094","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.12094"}},"official":{"repos":["AndssY/DeMa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/optimistic-query-routing-in-clustering-based","slug":"optimistic-query-routing-in-clustering-based","title":"Optimistic Query Routing in Clustering-based Approximate Maximum Inner Product Search","date":"2024-05-20","arxiv_id":"2405.12207","repositories_listed":1,"syntology":null},{"url":"/paper/what-hides-behind-unfairness-exploring","slug":"what-hides-behind-unfairness-exploring","title":"What Hides behind Unfairness? Exploring Dynamics Fairness in Reinforcement Learning","date":"2024-04-16","arxiv_id":"2404.10942","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/what-hides-behind-unfairness-exploring#ran","syntology_url":"https://syntology.ai/paper/2404.10942","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.10942"}},"official":{"repos":["familyld/insightfair"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-soft-actor-critic-with-global","slug":"multi-agent-soft-actor-critic-with-global","title":"Multi-Agent Soft Actor-Critic with Coordinated Loss for Autonomous Mobility-on-Demand Fleet Control","date":"2024-04-10","arxiv_id":"2404.06975","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-out-of-distribution-detection-for","slug":"rethinking-out-of-distribution-detection-for","title":"Rethinking Out-of-Distribution Detection for Reinforcement Learning: Advancing Methods for Evaluation and Detection","date":"2024-04-10","arxiv_id":"2404.07099","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rethinking-out-of-distribution-detection-for#ran","syntology_url":"https://syntology.ai/paper/2404.07099","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07099"}},"official":{"repos":["linasnas/dexter"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sequential-decision-making-with-expert","slug":"sequential-decision-making-with-expert","title":"Sequential Decision Making with Expert Demonstrations under Unobserved Heterogeneity","date":"2024-04-10","arxiv_id":"2404.07266","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-personalized-1","slug":"deep-reinforcement-learning-for-personalized-1","title":"Deep Reinforcement Learning for Personalized Diagnostic Decision Pathways Using Electronic Health Records: A Comparative Study on Anemia and Systemic Lupus Erythematosus","date":"2024-04-09","arxiv_id":"2404.05913","repositories_listed":1,"syntology":null},{"url":"/paper/decision-mamba-reinforcement-learning-via","slug":"decision-mamba-reinforcement-learning-via","title":"Decision Mamba: Reinforcement Learning via Sequence Modeling with Selective State Spaces","date":"2024-03-29","arxiv_id":"2403.19925","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/decision-mamba-reinforcement-learning-via#ran","syntology_url":"https://syntology.ai/paper/2403.19925","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19925"}},"official":{"repos":["toshihiro-ota/decision-mamba"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mixed-initiative-human-robot-teaming-under","slug":"mixed-initiative-human-robot-teaming-under","title":"Mixed-Initiative Human-Robot Teaming under Suboptimality with Online Bayesian Adaptation","date":"2024-03-24","arxiv_id":"2403.16178","repositories_listed":1,"syntology":null},{"url":"/paper/offline-imitation-of-badminton-player","slug":"offline-imitation-of-badminton-player","title":"Offline Imitation of Badminton Player Behavior via Experiential Contexts and Brownian Motion","date":"2024-03-19","arxiv_id":"2403.12406","repositories_listed":1,"syntology":null},{"url":"/paper/reinforced-sequential-decision-making-for","slug":"reinforced-sequential-decision-making-for","title":"Reinforced Sequential Decision-Making for Sepsis Treatment: The POSNEGDM Framework with Mortality Classifier and Transformer","date":"2024-03-12","arxiv_id":"2403.07309","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reinforced-sequential-decision-making-for#ran","syntology_url":"https://syntology.ai/paper/2403.07309","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07309"}},"official":{"repos":["dipeshtamboli/posnegdm-reinforced-sequential-decision-making-for-sepsis-treatment"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/trad-enhancing-llm-agents-with-step-wise","slug":"trad-enhancing-llm-agents-with-step-wise","title":"TRAD: Enhancing LLM Agents with Step-Wise Thought Retrieval and Aligned Decision","date":"2024-03-10","arxiv_id":"2403.06221","repositories_listed":1,"syntology":null},{"url":"/paper/how-can-llm-guide-rl-a-value-based-approach","slug":"how-can-llm-guide-rl-a-value-based-approach","title":"How Can LLM Guide RL? A Value-Based Approach","date":"2024-02-25","arxiv_id":"2402.16181","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/how-can-llm-guide-rl-a-value-based-approach#ran","syntology_url":"https://syntology.ai/paper/2402.16181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16181"}},"official":{"repos":["agentification/language-integrated-vi"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-design-for-justifiable-sequential","slug":"reward-design-for-justifiable-sequential","title":"Reward Design for Justifiable Sequential Decision-Making","date":"2024-02-24","arxiv_id":"2402.15826","repositories_listed":1,"syntology":null},{"url":"/paper/deep-generative-models-for-offline-policy","slug":"deep-generative-models-for-offline-policy","title":"Deep Generative Models for Offline Policy Learning: Tutorial, Survey, and Perspectives on Future Directions","date":"2024-02-21","arxiv_id":"2402.13777","repositories_listed":1,"syntology":null},{"url":"/paper/prise-learning-temporal-action-abstractions","slug":"prise-learning-temporal-action-abstractions","title":"PRISE: LLM-Style Sequence Compression for Learning Temporal Action Abstractions in Control","date":"2024-02-16","arxiv_id":"2402.10450","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/prise-learning-temporal-action-abstractions#ran","syntology_url":"https://syntology.ai/paper/2402.10450","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10450"}},"official":{"repos":["frankzheng2022/prise"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/jack-of-all-trades-master-of-some-a-multi","slug":"jack-of-all-trades-master-of-some-a-multi","title":"Jack of All Trades, Master of Some, a Multi-Purpose Transformer Agent","date":"2024-02-15","arxiv_id":"2402.09844","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/jack-of-all-trades-master-of-some-a-multi#ran","syntology_url":"https://syntology.ai/paper/2402.09844","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09844"}},"official":{"repos":["huggingface/jat"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/epistemic-exploration-for-generalizable","slug":"epistemic-exploration-for-generalizable","title":"Epistemic Exploration for Generalizable Planning and Learning in Non-Stationary Settings","date":"2024-02-13","arxiv_id":"2402.08145","repositories_listed":1,"syntology":null},{"url":"/paper/noise-adaptive-confidence-sets-for-linear","slug":"noise-adaptive-confidence-sets-for-linear","title":"Noise-Adaptive Confidence Sets for Linear Bandits and Application to Bayesian Optimization","date":"2024-02-12","arxiv_id":"2402.07341","repositories_listed":1,"syntology":{"n":15,"n_ran":15,"n_constructed":0,"n_ran_checked":12,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/noise-adaptive-confidence-sets-for-linear#ran","syntology_url":"https://syntology.ai/paper/2402.07341","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07341"}},"official":{"repos":["jungtaekkim/losan-lofav"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/premier-taco-pretraining-multitask","slug":"premier-taco-pretraining-multitask","title":"Premier-TACO is a Few-Shot Policy Learner: Pretraining Multitask Representation via Temporal Action-Driven Contrastive Loss","date":"2024-02-09","arxiv_id":"2402.06187","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/premier-taco-pretraining-multitask#ran","syntology_url":"https://syntology.ai/paper/2402.06187","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.06187"}},"official":{"repos":["premiertaco/premier-taco"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/logical-specifications-guided-dynamic-task","slug":"logical-specifications-guided-dynamic-task","title":"Logical Specifications-guided Dynamic Task Sampling for Reinforcement Learning Agents","date":"2024-02-06","arxiv_id":"2402.03678","repositories_listed":1,"syntology":null},{"url":"/paper/skill-set-optimization-reinforcing-language","slug":"skill-set-optimization-reinforcing-language","title":"Skill Set Optimization: Reinforcing Language Model Behavior via Transferable Skills","date":"2024-02-05","arxiv_id":"2402.03244","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/skill-set-optimization-reinforcing-language#ran","syntology_url":"https://syntology.ai/paper/2402.03244","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03244"}},"official":{"repos":["allenai/sso"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vertical-symbolic-regression-via-deep-policy","slug":"vertical-symbolic-regression-via-deep-policy","title":"Vertical Symbolic Regression via Deep Policy Gradient","date":"2024-02-01","arxiv_id":"2402.00254","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vertical-symbolic-regression-via-deep-policy#ran","syntology_url":"https://syntology.ai/paper/2402.00254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.00254"}},"official":{"repos":["jiangnanhugo/vsr-dpg"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/layered-and-staged-monte-carlo-tree-search","slug":"layered-and-staged-monte-carlo-tree-search","title":"Layered and Staged Monte Carlo Tree Search for SMT Strategy Synthesis","date":"2024-01-30","arxiv_id":"2401.17159","repositories_listed":1,"syntology":null},{"url":"/paper/long-term-fair-decision-making-through-deep","slug":"long-term-fair-decision-making-through-deep","title":"Long-Term Fair Decision Making through Deep Generative Models","date":"2024-01-20","arxiv_id":"2401.11288","repositories_listed":1,"syntology":null},{"url":"/paper/learning-non-myopic-power-allocation-in","slug":"learning-non-myopic-power-allocation-in","title":"Learning Non-myopic Power Allocation in Constrained Scenarios","date":"2024-01-18","arxiv_id":"2401.10297","repositories_listed":1,"syntology":null},{"url":"/paper/delf-designing-learning-environments-with","slug":"delf-designing-learning-environments-with","title":"DeLF: Designing Learning Environments with Foundation Models","date":"2024-01-17","arxiv_id":"2401.08936","repositories_listed":1,"syntology":null},{"url":"/paper/decision-making-in-non-stationary-1","slug":"decision-making-in-non-stationary-1","title":"Decision Making in Non-Stationary Environments with Policy-Augmented Search","date":"2024-01-06","arxiv_id":"2401.03197","repositories_listed":1,"syntology":null},{"url":"/paper/act-as-you-learn-adaptive-decision-making-in","slug":"act-as-you-learn-adaptive-decision-making-in","title":"Act as You Learn: Adaptive Decision-Making in Non-Stationary Markov Decision Processes","date":"2024-01-03","arxiv_id":"2401.01841","repositories_listed":1,"syntology":null},{"url":"/paper/harnessing-the-power-of-federated-learning-in","slug":"harnessing-the-power-of-federated-learning-in","title":"Harnessing the Power of Federated Learning in Federated Contextual Bandits","date":"2023-12-26","arxiv_id":"2312.16341","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/harnessing-the-power-of-federated-learning-in#ran","syntology_url":"https://syntology.ai/paper/2312.16341","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.16341"}},"official":{"repos":["shengroup/fedigw"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/risk-sensitive-stochastic-optimal-control-as","slug":"risk-sensitive-stochastic-optimal-control-as","title":"Risk-Sensitive Stochastic Optimal Control as Rao-Blackwellized Markovian Score Climbing","date":"2023-12-21","arxiv_id":"2312.14000","repositories_listed":1,"syntology":{"n":11,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":11,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/risk-sensitive-stochastic-optimal-control-as#ran","syntology_url":"https://syntology.ai/paper/2312.14000","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14000"}},"official":{"repos":["hanyas/psoc"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/solving-long-run-average-reward-robust-mdps","slug":"solving-long-run-average-reward-robust-mdps","title":"Solving Long-run Average Reward Robust MDPs via Stochastic Games","date":"2023-12-21","arxiv_id":"2312.13912","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/solving-long-run-average-reward-robust-mdps#ran","syntology_url":"https://syntology.ai/paper/2312.13912","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.13912"}},"official":{"repos":["mehrdad76/rmdp-lra"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/parameterized-projected-bellman-operator","slug":"parameterized-projected-bellman-operator","title":"Parameterized Projected Bellman Operator","date":"2023-12-20","arxiv_id":"2312.12869","repositories_listed":1,"syntology":null},{"url":"/paper/llm-ark-knowledge-graph-reasoning-using-large","slug":"llm-ark-knowledge-graph-reasoning-using-large","title":"Evaluating and Enhancing Large Language Models for Conversational Reasoning on Knowledge Graphs","date":"2023-12-18","arxiv_id":"2312.11282","repositories_listed":1,"syntology":null},{"url":"/paper/robust-active-measuring-under-model","slug":"robust-active-measuring-under-model","title":"Robust Active Measuring under Model Uncertainty","date":"2023-12-18","arxiv_id":"2312.11227","repositories_listed":1,"syntology":null},{"url":"/paper/risk-aware-continuous-control-with-neural","slug":"risk-aware-continuous-control-with-neural","title":"Risk-Aware Continuous Control with Neural Contextual Bandits","date":"2023-12-15","arxiv_id":"2312.09961","repositories_listed":1,"syntology":null},{"url":"/paper/llf-bench-benchmark-for-interactive-learning","slug":"llf-bench-benchmark-for-interactive-learning","title":"LLF-Bench: Benchmark for Interactive Learning from Language Feedback","date":"2023-12-11","arxiv_id":"2312.06853","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llf-bench-benchmark-for-interactive-learning#ran","syntology_url":"https://syntology.ai/paper/2312.06853","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06853"}},"official":null}},{"url":"/paper/online-decision-making-with-history-average","slug":"online-decision-making-with-history-average","title":"Online Decision Making with History-Average Dependent Costs (Extended)","date":"2023-12-11","arxiv_id":"2312.06641","repositories_listed":1,"syntology":null},{"url":"/paper/remembering-to-be-fair-on-non-markovian","slug":"remembering-to-be-fair-on-non-markovian","title":"Remembering to Be Fair: Non-Markovian Fairness in Sequential Decision Making","date":"2023-12-08","arxiv_id":"2312.04772","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/remembering-to-be-fair-on-non-markovian#ran","syntology_url":"https://syntology.ai/paper/2312.04772","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.04772"}},"official":{"repos":["praal/remembering-to-be-fair"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/can-language-agents-be-alternatives-to-ppo-a","slug":"can-language-agents-be-alternatives-to-ppo-a","title":"Can language agents be alternatives to PPO? A Preliminary Empirical Study On OpenAI Gym","date":"2023-12-06","arxiv_id":"2312.03290","repositories_listed":1,"syntology":null},{"url":"/paper/generalization-to-new-sequential-decision","slug":"generalization-to-new-sequential-decision","title":"Generalization to New Sequential Decision Making Tasks with In-Context Learning","date":"2023-12-06","arxiv_id":"2312.03801","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalization-to-new-sequential-decision#ran","syntology_url":"https://syntology.ai/paper/2312.03801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03801"}},"official":null}},{"url":"/paper/learning-curricula-in-open-ended-worlds","slug":"learning-curricula-in-open-ended-worlds","title":"Learning Curricula in Open-Ended Worlds","date":"2023-12-03","arxiv_id":"2312.03126","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-curricula-in-open-ended-worlds#ran","syntology_url":"https://syntology.ai/paper/2312.03126","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03126"}},"official":{"repos":["facebookresearch/dcd"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/token-recycling-for-efficient-sequential","slug":"token-recycling-for-efficient-sequential","title":"TORE: Token Recycling in Vision Transformers for Efficient Active Visual Exploration","date":"2023-11-26","arxiv_id":"2311.15335","repositories_listed":1,"syntology":null},{"url":"/paper/learning-dynamic-selection-and-pricing-of-out","slug":"learning-dynamic-selection-and-pricing-of-out","title":"Learning Dynamic Selection and Pricing of Out-of-Home Deliveries","date":"2023-11-23","arxiv_id":"2311.13983","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-model-is-a-good-policy-teacher","slug":"large-language-model-is-a-good-policy-teacher","title":"Large Language Model as a Policy Teacher for Training Reinforcement Learning Agents","date":"2023-11-22","arxiv_id":"2311.13373","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-symbolic-policy-learning-with","slug":"efficient-symbolic-policy-learning-with","title":"Efficient Symbolic Policy Learning with Differentiable Symbolic Expression","date":"2023-11-02","arxiv_id":"2311.02104","repositories_listed":1,"syntology":null},{"url":"/paper/free-from-bellman-completeness-trajectory","slug":"free-from-bellman-completeness-trajectory","title":"Free from Bellman Completeness: Trajectory Stitching via Model-based Return-conditioned Supervised Learning","date":"2023-10-30","arxiv_id":"2310.19308","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/free-from-bellman-completeness-trajectory#ran","syntology_url":"https://syntology.ai/paper/2310.19308","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19308"}},"official":{"repos":["zhaoyizhou1123/mbrcsl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/eureka-human-level-reward-design-via-coding","slug":"eureka-human-level-reward-design-via-coding","title":"Eureka: Human-Level Reward Design via Coding Large Language Models","date":"2023-10-19","arxiv_id":"2310.12931","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/eureka-human-level-reward-design-via-coding#ran","syntology_url":"https://syntology.ai/paper/2310.12931","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12931"}},"official":{"repos":["eureka-research/Eureka"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"url":"/paper/agent-specific-effects","slug":"agent-specific-effects","title":"Agent-Specific Effects: A Causal Effect Propagation Analysis in Multi-Agent MDPs","date":"2023-10-17","arxiv_id":"2310.11334","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/agent-specific-effects#ran","syntology_url":"https://syntology.ai/paper/2310.11334","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.11334"}},"official":{"repos":["stelios30/agent-specific-effects"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/imitation-learning-from-purified","slug":"imitation-learning-from-purified","title":"Imitation Learning from Purified Demonstrations","date":"2023-10-11","arxiv_id":"2310.07143","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":2,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imitation-learning-from-purified#ran","syntology_url":"https://syntology.ai/paper/2310.07143","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.07143"}},"official":{"repos":["yunke-wang/dp-il"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/learning-to-reach-goals-via-diffusion","slug":"learning-to-reach-goals-via-diffusion","title":"Learning to Reach Goals via Diffusion","date":"2023-10-04","arxiv_id":"2310.02505","repositories_listed":1,"syntology":null},{"url":"/paper/trace-trajectory-counterfactual-explanation","slug":"trace-trajectory-counterfactual-explanation","title":"TraCE: Trajectory Counterfactual Explanation Scores","date":"2023-09-27","arxiv_id":"2309.15965","repositories_listed":1,"syntology":{"n":14,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/trace-trajectory-counterfactual-explanation#ran","syntology_url":"https://syntology.ai/paper/2309.15965","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.15965"}},"official":{"repos":["jeffnclark/trace"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/interactively-teaching-an-inverse","slug":"interactively-teaching-an-inverse","title":"Interactively Teaching an Inverse Reinforcement Learner with Limited Feedback","date":"2023-09-16","arxiv_id":"2309.09095","repositories_listed":1,"syntology":null},{"url":"/paper/improving-reinforcement-learning-training","slug":"improving-reinforcement-learning-training","title":"Improving Generalization in Reinforcement Learning Training Regimes for Social Robot Navigation","date":"2023-08-29","arxiv_id":"2308.14947","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-reinforcement-learning-training#ran","syntology_url":"https://syntology.ai/paper/2308.14947","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.14947"}},"official":{"repos":["raise-lab/soc-nav-training"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/simple-modification-of-the-upper-confidence","slug":"simple-modification-of-the-upper-confidence","title":"Simple Modification of the Upper Confidence Bound Algorithm by Generalized Weighted Averages","date":"2023-08-28","arxiv_id":"2308.14350","repositories_listed":1,"syntology":null},{"url":"/paper/out-of-the-cage-how-stochastic-parrots-win-in","slug":"out-of-the-cage-how-stochastic-parrots-win-in","title":"Out of the Cage: How Stochastic Parrots Win in Cyber Security Environments","date":"2023-08-23","arxiv_id":"2308.12086","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/out-of-the-cage-how-stochastic-parrots-win-in#ran","syntology_url":"https://syntology.ai/paper/2308.12086","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12086"}},"official":{"repos":["stratosphereips/netsecgame"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lagr-seq-language-guided-reinforcement","slug":"lagr-seq-language-guided-reinforcement","title":"LaGR-SEQ: Language-Guided Reinforcement Learning with Sample-Efficient Querying","date":"2023-08-21","arxiv_id":"2308.13542","repositories_listed":1,"syntology":null},{"url":"/paper/value-distributional-model-based","slug":"value-distributional-model-based","title":"Value-Distributional Model-Based Reinforcement Learning","date":"2023-08-12","arxiv_id":"2308.06590","repositories_listed":1,"syntology":null},{"url":"/paper/flare-fingerprinting-deep-reinforcement","slug":"flare-fingerprinting-deep-reinforcement","title":"FLARE: Fingerprinting Deep Reinforcement Learning Agents using Universal Adversarial Masks","date":"2023-07-27","arxiv_id":"2307.14751","repositories_listed":1,"syntology":null},{"url":"/paper/hindsight-dice-stable-credit-assignment-for","slug":"hindsight-dice-stable-credit-assignment-for","title":"Hindsight-DICE: Stable Credit Assignment for Deep Reinforcement Learning","date":"2023-07-21","arxiv_id":"2307.11897","repositories_listed":1,"syntology":null},{"url":"/paper/breadcrumbs-to-the-goal-goal-conditioned","slug":"breadcrumbs-to-the-goal-goal-conditioned","title":"Breadcrumbs to the Goal: Goal-Conditioned Exploration from Human-in-the-Loop Feedback","date":"2023-07-20","arxiv_id":"2307.11049","repositories_listed":1,"syntology":null},{"url":"/paper/online-learning-with-costly-features-in-non","slug":"online-learning-with-costly-features-in-non","title":"Online Learning with Costly Features in Non-stationary Environments","date":"2023-07-18","arxiv_id":"2307.09388","repositories_listed":1,"syntology":null},{"url":"/paper/pomdp-inference-and-robust-solution-via-deep","slug":"pomdp-inference-and-robust-solution-via-deep","title":"POMDP inference and robust solution via deep reinforcement learning: An application to railway optimal maintenance","date":"2023-07-16","arxiv_id":"2307.08082","repositories_listed":1,"syntology":null},{"url":"/paper/safe-reinforcement-learning-as-wasserstein","slug":"safe-reinforcement-learning-as-wasserstein","title":"Probabilistic Constrained Reinforcement Learning with Formal Interpretability","date":"2023-07-13","arxiv_id":"2307.07084","repositories_listed":1,"syntology":null},{"url":"/paper/containergym-a-real-world-reinforcement","slug":"containergym-a-real-world-reinforcement","title":"ContainerGym: A Real-World Reinforcement Learning Benchmark for Resource Allocation","date":"2023-07-06","arxiv_id":"2307.02991","repositories_listed":1,"syntology":null},{"url":"/paper/learning-non-markovian-decision-making-from","slug":"learning-non-markovian-decision-making-from","title":"Learning non-Markovian Decision-Making from State-only Sequences","date":"2023-06-27","arxiv_id":"2306.15156","repositories_listed":1,"syntology":null},{"url":"/paper/sampling-from-gaussian-process-posteriors-1","slug":"sampling-from-gaussian-process-posteriors-1","title":"Sampling from Gaussian Process Posteriors using Stochastic Gradient Descent","date":"2023-06-20","arxiv_id":"2306.11589","repositories_listed":1,"syntology":null},{"url":"/paper/simplified-temporal-consistency-reinforcement","slug":"simplified-temporal-consistency-reinforcement","title":"Simplified Temporal Consistency Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.09466","repositories_listed":1,"syntology":null},{"url":"/paper/skill-disentanglement-for-imitation-learning","slug":"skill-disentanglement-for-imitation-learning","title":"Skill Disentanglement for Imitation Learning from Suboptimal Demonstrations","date":"2023-06-13","arxiv_id":"2306.07919","repositories_listed":1,"syntology":null},{"url":"/paper/decision-stacks-flexible-reinforcement","slug":"decision-stacks-flexible-reinforcement","title":"Decision Stacks: Flexible Reinforcement Learning via Modular Generative Models","date":"2023-06-09","arxiv_id":"2306.06253","repositories_listed":1,"syntology":null},{"url":"/paper/autonomous-capability-assessment-of-black-box","slug":"autonomous-capability-assessment-of-black-box","title":"Autonomous Capability Assessment of Sequential Decision-Making Systems in Stochastic Settings (Extended Version)","date":"2023-06-07","arxiv_id":"2306.04806","repositories_listed":1,"syntology":null},{"url":"/paper/professional-basketball-player-behavior","slug":"professional-basketball-player-behavior","title":"PlayBest: Professional Basketball Player Behavior Synthesis via Planning with Diffusion","date":"2023-06-07","arxiv_id":"2306.04090","repositories_listed":1,"syntology":null},{"url":"/paper/enabling-efficient-interaction-between-an","slug":"enabling-efficient-interaction-between-an","title":"Enabling Intelligent Interactions between an Agent and an LLM: A Reinforcement Learning Approach","date":"2023-06-06","arxiv_id":"2306.03604","repositories_listed":1,"syntology":null},{"url":"/paper/finding-counterfactually-optimal-action","slug":"finding-counterfactually-optimal-action","title":"Finding Counterfactually Optimal Action Sequences in Continuous State Spaces","date":"2023-06-06","arxiv_id":"2306.03929","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/finding-counterfactually-optimal-action#ran","syntology_url":"https://syntology.ai/paper/2306.03929","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03929"}},"official":{"repos":["networks-learning/counterfactual-continuous-mdp"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/extracting-reward-functions-from-diffusion","slug":"extracting-reward-functions-from-diffusion","title":"Extracting Reward Functions from Diffusion Models","date":"2023-06-01","arxiv_id":"2306.01804","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/extracting-reward-functions-from-diffusion#ran","syntology_url":"https://syntology.ai/paper/2306.01804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01804"}},"official":{"repos":["FelipeNuti/diffusion-relative-rewards"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/steve-1-a-generative-model-for-text-to","slug":"steve-1-a-generative-model-for-text-to","title":"STEVE-1: A Generative Model for Text-to-Behavior in Minecraft","date":"2023-06-01","arxiv_id":"2306.00937","repositories_listed":1,"syntology":null},{"url":"/paper/modeling-adversarial-attack-on-pre-trained","slug":"modeling-adversarial-attack-on-pre-trained","title":"Modeling Adversarial Attack on Pre-trained Language Models as Sequential Decision Making","date":"2023-05-27","arxiv_id":"2305.17440","repositories_listed":1,"syntology":null},{"url":"/paper/adaplanner-adaptive-planning-from-feedback-1","slug":"adaplanner-adaptive-planning-from-feedback-1","title":"AdaPlanner: Adaptive Planning from Feedback with Language Models","date":"2023-05-26","arxiv_id":"2305.16653","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaplanner-adaptive-planning-from-feedback-1#ran","syntology_url":"https://syntology.ai/paper/2305.16653","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.16653"}},"official":{"repos":["haotiansun14/adaplanner"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/masked-trajectory-models-for-prediction","slug":"masked-trajectory-models-for-prediction","title":"Masked Trajectory Models for Prediction, Representation, and Control","date":"2023-05-04","arxiv_id":"2305.02968","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/masked-trajectory-models-for-prediction#ran","syntology_url":"https://syntology.ai/paper/2305.02968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.02968"}},"official":{"repos":["facebookresearch/mtm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/semi-infinitely-constrained-markov-decision","slug":"semi-infinitely-constrained-markov-decision","title":"Semi-Infinitely Constrained Markov Decision Processes and Efficient Reinforcement Learning","date":"2023-04-29","arxiv_id":"2305.00254","repositories_listed":1,"syntology":null},{"url":"/paper/x-rlflow-graph-reinforcement-learning-for","slug":"x-rlflow-graph-reinforcement-learning-for","title":"X-RLflow: Graph Reinforcement Learning for Neural Network Subgraphs Transformation","date":"2023-04-28","arxiv_id":"2304.14698","repositories_listed":1,"syntology":null},{"url":"/paper/distance-weighted-supervised-learning-for","slug":"distance-weighted-supervised-learning-for","title":"Distance Weighted Supervised Learning for Offline Interaction Data","date":"2023-04-26","arxiv_id":"2304.13774","repositories_listed":1,"syntology":{"n":27,"n_ran":14,"n_constructed":0,"n_ran_checked":4,"n_instrument":10,"n_unverified":13,"n_honours":2,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 1 violated, 1 with no contract checked; 10 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/distance-weighted-supervised-learning-for#ran","syntology_url":"https://syntology.ai/paper/2304.13774","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.13774"}},"official":null}},{"url":"/paper/temporl-laser-pulse-temporal-shape","slug":"temporl-laser-pulse-temporal-shape","title":"TempoRL: laser pulse temporal shape optimization with Deep Reinforcement Learning","date":"2023-04-20","arxiv_id":"2304.12187","repositories_listed":1,"syntology":null},{"url":"/paper/promises-and-pitfalls-of-the-linearized","slug":"promises-and-pitfalls-of-the-linearized","title":"Promises and Pitfalls of the Linearized Laplace in Bayesian Optimization","date":"2023-04-17","arxiv_id":"2304.08309","repositories_listed":1,"syntology":null},{"url":"/paper/automaton-guided-curriculum-generation-for","slug":"automaton-guided-curriculum-generation-for","title":"Automaton-Guided Curriculum Generation for Reinforcement Learning Agents","date":"2023-04-11","arxiv_id":"2304.05271","repositories_listed":1,"syntology":null},{"url":"/paper/did-we-personalize-assessing-personalization","slug":"did-we-personalize-assessing-personalization","title":"Did we personalize? Assessing personalization by an online reinforcement learning algorithm using resampling","date":"2023-04-11","arxiv_id":"2304.05365","repositories_listed":1,"syntology":null},{"url":"/paper/mahalo-unifying-offline-reinforcement","slug":"mahalo-unifying-offline-reinforcement","title":"MAHALO: Unifying Offline Reinforcement Learning and Imitation Learning from Observations","date":"2023-03-30","arxiv_id":"2303.17156","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mahalo-unifying-offline-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2303.17156","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17156"}},"official":{"repos":["anqili/mahalo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/probabilistic-inverse-optimal-control-for-non","slug":"probabilistic-inverse-optimal-control-for-non","title":"Probabilistic inverse optimal control for non-linear partially observable systems disentangles perceptual uncertainty and behavioral costs","date":"2023-03-29","arxiv_id":"2303.16698","repositories_listed":1,"syntology":null},{"url":"/paper/merging-decision-transformers-weight","slug":"merging-decision-transformers-weight","title":"Merging Decision Transformers: Weight Averaging for Forming Multi-Task Policies","date":"2023-03-14","arxiv_id":"2303.07551","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/merging-decision-transformers-weight#ran","syntology_url":"https://syntology.ai/paper/2303.07551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.07551"}},"official":{"repos":["daniellawson9999/merging-decision-transformers"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/communication-efficient-collaborative","slug":"communication-efficient-collaborative","title":"Flooding with Absorption: An Efficient Protocol for Heterogeneous Bandits over Complex Networks","date":"2023-03-09","arxiv_id":"2303.05445","repositories_listed":1,"syntology":null},{"url":"/paper/population-based-evaluation-in-repeated-rock","slug":"population-based-evaluation-in-repeated-rock","title":"Population-based Evaluation in Repeated Rock-Paper-Scissors as a Benchmark for Multiagent Reinforcement Learning","date":"2023-03-02","arxiv_id":"2303.03196","repositories_listed":1,"syntology":null},{"url":"/paper/causal-social-explanations-for-stochastic","slug":"causal-social-explanations-for-stochastic","title":"Causal Explanations for Sequential Decision-Making in Multi-Agent Systems","date":"2023-02-21","arxiv_id":"2302.10809","repositories_listed":1,"syntology":null},{"url":"/paper/minimax-bayes-reinforcement-learning","slug":"minimax-bayes-reinforcement-learning","title":"Minimax-Bayes Reinforcement Learning","date":"2023-02-21","arxiv_id":"2302.10831","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-simplex-balancing-safety-and","slug":"dynamic-simplex-balancing-safety-and","title":"Dynamic Simplex: Balancing Safety and Performance in Autonomous Cyber Physical Systems","date":"2023-02-20","arxiv_id":"2302.09750","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-simplex-balancing-safety-and#ran","syntology_url":"https://syntology.ai/paper/2302.09750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.09750"}},"official":{"repos":["BaitingLuo/Dynamic_Simplex"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/best-arm-identification-for-stochastic-rising","slug":"best-arm-identification-for-stochastic-rising","title":"Best Arm Identification for Stochastic Rising Bandits","date":"2023-02-15","arxiv_id":"2302.07510","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/best-arm-identification-for-stochastic-rising#ran","syntology_url":"https://syntology.ai/paper/2302.07510","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.07510"}},"official":{"repos":["montenegroalessandro/bestarmidsrb"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"27b966dfa14090c60be0df37953f6335560610aa2dbfe09e3d2e8814488479d7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}