{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/sequential-decision-making/papers/ran/1","list_of":"/task/sequential-decision-making","task":"Sequential Decision Making","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":2,"rows_per_page":100,"rows":[1,100],"of":107,"counts":{"archive_papers_tagged":1210,"with_a_code_link":351,"where_syntology_ran_a_sample":107,"not_listed_spam_title":0,"listed":1210,"listed_where_code_ran":107,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":90,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":90,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/sequential-decision-making/papers/ran/1","prev":null,"next":"/task/sequential-decision-making/papers/ran/2","papers":[{"url":"/paper/flow-based-single-step-completion-for","slug":"flow-based-single-step-completion-for","title":"Flow-Based Single-Step Completion for Efficient and Expressive Policy Learning","date":"2025-06-26","arxiv_id":"2506.21427","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/flow-based-single-step-completion-for#ran","syntology_url":"https://syntology.ai/paper/2506.21427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.21427"}},"official":null}},{"url":"/paper/textatari-100k-frames-game-playing-with","slug":"textatari-100k-frames-game-playing-with","title":"TextAtari: 100K Frames Game Playing with Language Agents","date":"2025-06-04","arxiv_id":"2506.04098","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/textatari-100k-frames-game-playing-with#ran","syntology_url":"https://syntology.ai/paper/2506.04098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.04098"}},"official":{"repos":["Lww007/Text-Atari-Agents"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/web-shepherd-advancing-prms-for-reinforcing","slug":"web-shepherd-advancing-prms-for-reinforcing","title":"Web-Shepherd: Advancing PRMs for Reinforcing Web Agents","date":"2025-05-21","arxiv_id":"2505.15277","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/web-shepherd-advancing-prms-for-reinforcing#ran","syntology_url":"https://syntology.ai/paper/2505.15277","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15277"}},"official":{"repos":["kyle8581/Web-Shepherd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-combinatorial-1","slug":"reinforcement-learning-with-combinatorial-1","title":"Reinforcement learning with combinatorial actions for coupled restless bandits","date":"2025-03-01","arxiv_id":"2503.01919","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-combinatorial-1#ran","syntology_url":"https://syntology.ai/paper/2503.01919","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.01919"}},"official":{"repos":["lily-x/combinatorial-rmab"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-solve-the-min-max-mixed-shelves","slug":"learning-to-solve-the-min-max-mixed-shelves","title":"Learning to Solve the Min-Max Mixed-Shelves Picker-Routing Problem via Hierarchical and Parallel Decoding","date":"2025-02-14","arxiv_id":"2502.10233","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-solve-the-min-max-mixed-shelves#ran","syntology_url":"https://syntology.ai/paper/2502.10233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.10233"}},"official":{"repos":["ltluttmann/marl4msprp"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/minsstudio-a-streamlined-package-for","slug":"minsstudio-a-streamlined-package-for","title":"MineStudio: A Streamlined Package for Minecraft AI Agent Development","date":"2024-12-24","arxiv_id":"2412.18293","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/minsstudio-a-streamlined-package-for#ran","syntology_url":"https://syntology.ai/paper/2412.18293","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.18293"}},"official":{"repos":["craftjarvis/minestudio"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pagerank-bandits-for-link-prediction","slug":"pagerank-bandits-for-link-prediction","title":"PageRank Bandits for Link Prediction","date":"2024-11-03","arxiv_id":"2411.01410","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pagerank-bandits-for-link-prediction#ran","syntology_url":"https://syntology.ai/paper/2411.01410","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.01410"}},"official":{"repos":["jiaruzouu/prb"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/logicity-advancing-neuro-symbolic-ai-with","slug":"logicity-advancing-neuro-symbolic-ai-with","title":"LogiCity: Advancing Neuro-Symbolic AI with Abstract Urban Simulation","date":"2024-11-01","arxiv_id":"2411.00773","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/logicity-advancing-neuro-symbolic-ai-with#ran","syntology_url":"https://syntology.ai/paper/2411.00773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00773"}},"official":{"repos":["Jaraxxus-Me/LogiCity"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-versatile-skills-with-curriculum","slug":"learning-versatile-skills-with-curriculum","title":"Learning Versatile Skills with Curriculum Masking","date":"2024-10-23","arxiv_id":"2410.17744","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-versatile-skills-with-curriculum#ran","syntology_url":"https://syntology.ai/paper/2410.17744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17744"}},"official":{"repos":["yaotang23/currmask"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dataenvgym-data-generation-agents-in-teacher","slug":"dataenvgym-data-generation-agents-in-teacher","title":"DataEnvGym: Data Generation Agents in Teacher Environments with Student Feedback","date":"2024-10-08","arxiv_id":"2410.06215","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dataenvgym-data-generation-agents-in-teacher#ran","syntology_url":"https://syntology.ai/paper/2410.06215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.06215"}},"official":{"repos":["codezakh/dataenvgym"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dopl-direct-online-preference-learning-for","slug":"dopl-direct-online-preference-learning-for","title":"DOPL: Direct Online Preference Learning for Restless Bandits with Preference Feedback","date":"2024-10-07","arxiv_id":"2410.05527","repositories_listed":0,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dopl-direct-online-preference-learning-for#ran","syntology_url":"https://syntology.ai/paper/2410.05527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05527"}},"official":null}},{"url":"/paper/learning-a-fast-mixing-exogenous-block-mdp","slug":"learning-a-fast-mixing-exogenous-block-mdp","title":"Learning a Fast Mixing Exogenous Block MDP using a Single Trajectory","date":"2024-10-03","arxiv_id":"2410.03016","repositories_listed":1,"syntology":{"n":6,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/learning-a-fast-mixing-exogenous-block-mdp#ran","syntology_url":"https://syntology.ai/paper/2410.03016","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03016"}},"official":{"repos":["midi-lab/steel"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-teachers-for-amortized-samplers","slug":"adaptive-teachers-for-amortized-samplers","title":"Adaptive teachers for amortized samplers","date":"2024-10-02","arxiv_id":"2410.01432","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":1,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptive-teachers-for-amortized-samplers#ran","syntology_url":"https://syntology.ai/paper/2410.01432","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01432"}},"official":{"repos":["alstn12088/adaptive-teacher"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-03195","slug":"2408-03195","title":"RELIEF: Reinforcement Learning Empowered Graph Feature Prompt Tuning","date":"2024-08-06","arxiv_id":"2408.03195","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":15,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/2408-03195#ran","syntology_url":"https://syntology.ai/paper/2408.03195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.03195"}},"official":{"repos":["JasonZhujp/RELIEF"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-foundation-models-for-online","slug":"adaptive-foundation-models-for-online","title":"Scalable Exploration via Ensemble++","date":"2024-07-18","arxiv_id":"2407.13195","repositories_listed":2,"syntology":{"n":8,"n_ran":5,"n_constructed":1,"n_ran_checked":3,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":8,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/adaptive-foundation-models-for-online#ran","syntology_url":"https://syntology.ai/paper/2407.13195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.13195"}},"official":{"repos":["szrlee/GPT-HyperAgent","szrlee/ensemble_plus_plus"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/last-iterate-global-convergence-of-policy","slug":"last-iterate-global-convergence-of-policy","title":"Last-Iterate Global Convergence of Policy Gradients for Constrained Reinforcement Learning","date":"2024-07-15","arxiv_id":"2407.10775","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/last-iterate-global-convergence-of-policy#ran","syntology_url":"https://syntology.ai/paper/2407.10775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.10775"}},"official":null}},{"url":"/paper/preserving-the-privacy-of-reward-functions-in","slug":"preserving-the-privacy-of-reward-functions-in","title":"Preserving the Privacy of Reward Functions in MDPs through Deception","date":"2024-07-13","arxiv_id":"2407.09809","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/preserving-the-privacy-of-reward-functions-in#ran","syntology_url":"https://syntology.ai/paper/2407.09809","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09809"}},"official":{"repos":["shshnkreddy/deceptiverl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/macrohft-memory-augmented-context-aware","slug":"macrohft-memory-augmented-context-aware","title":"MacroHFT: Memory Augmented Context-aware Reinforcement Learning On High Frequency Trading","date":"2024-06-20","arxiv_id":"2406.14537","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/macrohft-memory-augmented-context-aware#ran","syntology_url":"https://syntology.ai/paper/2406.14537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14537"}},"official":{"repos":["ZONG0004/MacroHFT"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pursuing-overall-welfare-in-federated","slug":"pursuing-overall-welfare-in-federated","title":"Pursuing Overall Welfare in Federated Learning through Sequential Decision Making","date":"2024-05-31","arxiv_id":"2405.20821","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pursuing-overall-welfare-in-federated#ran","syntology_url":"https://syntology.ai/paper/2405.20821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20821"}},"official":{"repos":["vaseline555/aaggff"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-machines-for-deep-rl-in-noisy-and","slug":"reward-machines-for-deep-rl-in-noisy-and","title":"Reward Machines for Deep RL in Noisy and Uncertain Environments","date":"2024-05-31","arxiv_id":"2406.00120","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/reward-machines-for-deep-rl-in-noisy-and#ran","syntology_url":"https://syntology.ai/paper/2406.00120","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00120"}},"official":{"repos":["andrewli77/reward-machines-noisy-environments"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/is-mamba-compatible-with-trajectory","slug":"is-mamba-compatible-with-trajectory","title":"Is Mamba Compatible with Trajectory Optimization in Offline Reinforcement Learning?","date":"2024-05-20","arxiv_id":"2405.12094","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/is-mamba-compatible-with-trajectory#ran","syntology_url":"https://syntology.ai/paper/2405.12094","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.12094"}},"official":{"repos":["AndssY/DeMa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/what-hides-behind-unfairness-exploring","slug":"what-hides-behind-unfairness-exploring","title":"What Hides behind Unfairness? Exploring Dynamics Fairness in Reinforcement Learning","date":"2024-04-16","arxiv_id":"2404.10942","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/what-hides-behind-unfairness-exploring#ran","syntology_url":"https://syntology.ai/paper/2404.10942","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.10942"}},"official":{"repos":["familyld/insightfair"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-out-of-distribution-detection-for","slug":"rethinking-out-of-distribution-detection-for","title":"Rethinking Out-of-Distribution Detection for Reinforcement Learning: Advancing Methods for Evaluation and Detection","date":"2024-04-10","arxiv_id":"2404.07099","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rethinking-out-of-distribution-detection-for#ran","syntology_url":"https://syntology.ai/paper/2404.07099","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07099"}},"official":{"repos":["linasnas/dexter"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/decision-mamba-reinforcement-learning-via","slug":"decision-mamba-reinforcement-learning-via","title":"Decision Mamba: Reinforcement Learning via Sequence Modeling with Selective State Spaces","date":"2024-03-29","arxiv_id":"2403.19925","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/decision-mamba-reinforcement-learning-via#ran","syntology_url":"https://syntology.ai/paper/2403.19925","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19925"}},"official":{"repos":["toshihiro-ota/decision-mamba"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforced-sequential-decision-making-for","slug":"reinforced-sequential-decision-making-for","title":"Reinforced Sequential Decision-Making for Sepsis Treatment: The POSNEGDM Framework with Mortality Classifier and Transformer","date":"2024-03-12","arxiv_id":"2403.07309","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reinforced-sequential-decision-making-for#ran","syntology_url":"https://syntology.ai/paper/2403.07309","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07309"}},"official":{"repos":["dipeshtamboli/posnegdm-reinforced-sequential-decision-making-for-sepsis-treatment"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/how-can-llm-guide-rl-a-value-based-approach","slug":"how-can-llm-guide-rl-a-value-based-approach","title":"How Can LLM Guide RL? A Value-Based Approach","date":"2024-02-25","arxiv_id":"2402.16181","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/how-can-llm-guide-rl-a-value-based-approach#ran","syntology_url":"https://syntology.ai/paper/2402.16181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16181"}},"official":{"repos":["agentification/language-integrated-vi"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/prise-learning-temporal-action-abstractions","slug":"prise-learning-temporal-action-abstractions","title":"PRISE: LLM-Style Sequence Compression for Learning Temporal Action Abstractions in Control","date":"2024-02-16","arxiv_id":"2402.10450","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/prise-learning-temporal-action-abstractions#ran","syntology_url":"https://syntology.ai/paper/2402.10450","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10450"}},"official":{"repos":["frankzheng2022/prise"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/jack-of-all-trades-master-of-some-a-multi","slug":"jack-of-all-trades-master-of-some-a-multi","title":"Jack of All Trades, Master of Some, a Multi-Purpose Transformer Agent","date":"2024-02-15","arxiv_id":"2402.09844","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/jack-of-all-trades-master-of-some-a-multi#ran","syntology_url":"https://syntology.ai/paper/2402.09844","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09844"}},"official":{"repos":["huggingface/jat"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/noise-adaptive-confidence-sets-for-linear","slug":"noise-adaptive-confidence-sets-for-linear","title":"Noise-Adaptive Confidence Sets for Linear Bandits and Application to Bayesian Optimization","date":"2024-02-12","arxiv_id":"2402.07341","repositories_listed":1,"syntology":{"n":15,"n_ran":15,"n_constructed":0,"n_ran_checked":12,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/noise-adaptive-confidence-sets-for-linear#ran","syntology_url":"https://syntology.ai/paper/2402.07341","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07341"}},"official":{"repos":["jungtaekkim/losan-lofav"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/premier-taco-pretraining-multitask","slug":"premier-taco-pretraining-multitask","title":"Premier-TACO is a Few-Shot Policy Learner: Pretraining Multitask Representation via Temporal Action-Driven Contrastive Loss","date":"2024-02-09","arxiv_id":"2402.06187","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/premier-taco-pretraining-multitask#ran","syntology_url":"https://syntology.ai/paper/2402.06187","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.06187"}},"official":{"repos":["premiertaco/premier-taco"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/skill-set-optimization-reinforcing-language","slug":"skill-set-optimization-reinforcing-language","title":"Skill Set Optimization: Reinforcing Language Model Behavior via Transferable Skills","date":"2024-02-05","arxiv_id":"2402.03244","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/skill-set-optimization-reinforcing-language#ran","syntology_url":"https://syntology.ai/paper/2402.03244","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03244"}},"official":{"repos":["allenai/sso"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vertical-symbolic-regression-via-deep-policy","slug":"vertical-symbolic-regression-via-deep-policy","title":"Vertical Symbolic Regression via Deep Policy Gradient","date":"2024-02-01","arxiv_id":"2402.00254","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vertical-symbolic-regression-via-deep-policy#ran","syntology_url":"https://syntology.ai/paper/2402.00254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.00254"}},"official":{"repos":["jiangnanhugo/vsr-dpg"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-shot-reinforcement-learning-via-function","slug":"zero-shot-reinforcement-learning-via-function","title":"Zero-Shot Reinforcement Learning via Function Encoders","date":"2024-01-30","arxiv_id":"2401.17173","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/zero-shot-reinforcement-learning-via-function#ran","syntology_url":"https://syntology.ai/paper/2401.17173","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.17173"}},"official":{"repos":["anonymousresearcher5642/functionencoderrl","tyler-ingebrand/functionencoderrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/harnessing-the-power-of-federated-learning-in","slug":"harnessing-the-power-of-federated-learning-in","title":"Harnessing the Power of Federated Learning in Federated Contextual Bandits","date":"2023-12-26","arxiv_id":"2312.16341","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/harnessing-the-power-of-federated-learning-in#ran","syntology_url":"https://syntology.ai/paper/2312.16341","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.16341"}},"official":{"repos":["shengroup/fedigw"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/solving-long-run-average-reward-robust-mdps","slug":"solving-long-run-average-reward-robust-mdps","title":"Solving Long-run Average Reward Robust MDPs via Stochastic Games","date":"2023-12-21","arxiv_id":"2312.13912","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/solving-long-run-average-reward-robust-mdps#ran","syntology_url":"https://syntology.ai/paper/2312.13912","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.13912"}},"official":{"repos":["mehrdad76/rmdp-lra"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/risk-sensitive-stochastic-optimal-control-as","slug":"risk-sensitive-stochastic-optimal-control-as","title":"Risk-Sensitive Stochastic Optimal Control as Rao-Blackwellized Markovian Score Climbing","date":"2023-12-21","arxiv_id":"2312.14000","repositories_listed":1,"syntology":{"n":11,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":11,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/risk-sensitive-stochastic-optimal-control-as#ran","syntology_url":"https://syntology.ai/paper/2312.14000","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14000"}},"official":{"repos":["hanyas/psoc"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/llf-bench-benchmark-for-interactive-learning","slug":"llf-bench-benchmark-for-interactive-learning","title":"LLF-Bench: Benchmark for Interactive Learning from Language Feedback","date":"2023-12-11","arxiv_id":"2312.06853","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llf-bench-benchmark-for-interactive-learning#ran","syntology_url":"https://syntology.ai/paper/2312.06853","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06853"}},"official":null}},{"url":"/paper/generalization-to-new-sequential-decision","slug":"generalization-to-new-sequential-decision","title":"Generalization to New Sequential Decision Making Tasks with In-Context Learning","date":"2023-12-06","arxiv_id":"2312.03801","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalization-to-new-sequential-decision#ran","syntology_url":"https://syntology.ai/paper/2312.03801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03801"}},"official":null}},{"url":"/paper/learning-curricula-in-open-ended-worlds","slug":"learning-curricula-in-open-ended-worlds","title":"Learning Curricula in Open-Ended Worlds","date":"2023-12-03","arxiv_id":"2312.03126","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-curricula-in-open-ended-worlds#ran","syntology_url":"https://syntology.ai/paper/2312.03126","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03126"}},"official":{"repos":["facebookresearch/dcd"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/eureka-human-level-reward-design-via-coding","slug":"eureka-human-level-reward-design-via-coding","title":"Eureka: Human-Level Reward Design via Coding Large Language Models","date":"2023-10-19","arxiv_id":"2310.12931","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/eureka-human-level-reward-design-via-coding#ran","syntology_url":"https://syntology.ai/paper/2310.12931","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12931"}},"official":{"repos":["eureka-research/Eureka"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"url":"/paper/agent-specific-effects","slug":"agent-specific-effects","title":"Agent-Specific Effects: A Causal Effect Propagation Analysis in Multi-Agent MDPs","date":"2023-10-17","arxiv_id":"2310.11334","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/agent-specific-effects#ran","syntology_url":"https://syntology.ai/paper/2310.11334","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.11334"}},"official":{"repos":["stelios30/agent-specific-effects"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/imitation-learning-from-purified","slug":"imitation-learning-from-purified","title":"Imitation Learning from Purified Demonstrations","date":"2023-10-11","arxiv_id":"2310.07143","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":2,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imitation-learning-from-purified#ran","syntology_url":"https://syntology.ai/paper/2310.07143","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.07143"}},"official":{"repos":["yunke-wang/dp-il"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/trace-trajectory-counterfactual-explanation","slug":"trace-trajectory-counterfactual-explanation","title":"TraCE: Trajectory Counterfactual Explanation Scores","date":"2023-09-27","arxiv_id":"2309.15965","repositories_listed":1,"syntology":{"n":14,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/trace-trajectory-counterfactual-explanation#ran","syntology_url":"https://syntology.ai/paper/2309.15965","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.15965"}},"official":{"repos":["jeffnclark/trace"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-reinforcement-learning-training","slug":"improving-reinforcement-learning-training","title":"Improving Generalization in Reinforcement Learning Training Regimes for Social Robot Navigation","date":"2023-08-29","arxiv_id":"2308.14947","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-reinforcement-learning-training#ran","syntology_url":"https://syntology.ai/paper/2308.14947","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.14947"}},"official":{"repos":["raise-lab/soc-nav-training"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/out-of-the-cage-how-stochastic-parrots-win-in","slug":"out-of-the-cage-how-stochastic-parrots-win-in","title":"Out of the Cage: How Stochastic Parrots Win in Cyber Security Environments","date":"2023-08-23","arxiv_id":"2308.12086","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/out-of-the-cage-how-stochastic-parrots-win-in#ran","syntology_url":"https://syntology.ai/paper/2308.12086","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12086"}},"official":{"repos":["stratosphereips/netsecgame"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/finding-counterfactually-optimal-action","slug":"finding-counterfactually-optimal-action","title":"Finding Counterfactually Optimal Action Sequences in Continuous State Spaces","date":"2023-06-06","arxiv_id":"2306.03929","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/finding-counterfactually-optimal-action#ran","syntology_url":"https://syntology.ai/paper/2306.03929","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03929"}},"official":{"repos":["networks-learning/counterfactual-continuous-mdp"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/extracting-reward-functions-from-diffusion","slug":"extracting-reward-functions-from-diffusion","title":"Extracting Reward Functions from Diffusion Models","date":"2023-06-01","arxiv_id":"2306.01804","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/extracting-reward-functions-from-diffusion#ran","syntology_url":"https://syntology.ai/paper/2306.01804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01804"}},"official":{"repos":["FelipeNuti/diffusion-relative-rewards"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adaplanner-adaptive-planning-from-feedback-1","slug":"adaplanner-adaptive-planning-from-feedback-1","title":"AdaPlanner: Adaptive Planning from Feedback with Language Models","date":"2023-05-26","arxiv_id":"2305.16653","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaplanner-adaptive-planning-from-feedback-1#ran","syntology_url":"https://syntology.ai/paper/2305.16653","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.16653"}},"official":{"repos":["haotiansun14/adaplanner"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/masked-trajectory-models-for-prediction","slug":"masked-trajectory-models-for-prediction","title":"Masked Trajectory Models for Prediction, Representation, and Control","date":"2023-05-04","arxiv_id":"2305.02968","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/masked-trajectory-models-for-prediction#ran","syntology_url":"https://syntology.ai/paper/2305.02968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.02968"}},"official":{"repos":["facebookresearch/mtm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/distance-weighted-supervised-learning-for","slug":"distance-weighted-supervised-learning-for","title":"Distance Weighted Supervised Learning for Offline Interaction Data","date":"2023-04-26","arxiv_id":"2304.13774","repositories_listed":1,"syntology":{"n":27,"n_ran":14,"n_constructed":0,"n_ran_checked":4,"n_instrument":10,"n_unverified":13,"n_honours":2,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 1 violated, 1 with no contract checked; 10 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/distance-weighted-supervised-learning-for#ran","syntology_url":"https://syntology.ai/paper/2304.13774","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.13774"}},"official":null}},{"url":"/paper/mahalo-unifying-offline-reinforcement","slug":"mahalo-unifying-offline-reinforcement","title":"MAHALO: Unifying Offline Reinforcement Learning and Imitation Learning from Observations","date":"2023-03-30","arxiv_id":"2303.17156","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mahalo-unifying-offline-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2303.17156","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17156"}},"official":{"repos":["anqili/mahalo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reflexion-language-agents-with-verbal","slug":"reflexion-language-agents-with-verbal","title":"Reflexion: Language Agents with Verbal Reinforcement Learning","date":"2023-03-20","arxiv_id":"2303.11366","repositories_listed":5,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reflexion-language-agents-with-verbal#ran","syntology_url":"https://syntology.ai/paper/2303.11366","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11366"}},"official":{"repos":["noahshinn024/reflexion"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/merging-decision-transformers-weight","slug":"merging-decision-transformers-weight","title":"Merging Decision Transformers: Weight Averaging for Forming Multi-Task Policies","date":"2023-03-14","arxiv_id":"2303.07551","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/merging-decision-transformers-weight#ran","syntology_url":"https://syntology.ai/paper/2303.07551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.07551"}},"official":{"repos":["daniellawson9999/merging-decision-transformers"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-simplex-balancing-safety-and","slug":"dynamic-simplex-balancing-safety-and","title":"Dynamic Simplex: Balancing Safety and Performance in Autonomous Cyber Physical Systems","date":"2023-02-20","arxiv_id":"2302.09750","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-simplex-balancing-safety-and#ran","syntology_url":"https://syntology.ai/paper/2302.09750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.09750"}},"official":{"repos":["BaitingLuo/Dynamic_Simplex"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/trieste-efficiently-exploring-the-depths-of","slug":"trieste-efficiently-exploring-the-depths-of","title":"Trieste: Efficiently Exploring The Depths of Black-box Functions with TensorFlow","date":"2023-02-16","arxiv_id":"2302.08436","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trieste-efficiently-exploring-the-depths-of#ran","syntology_url":"https://syntology.ai/paper/2302.08436","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.08436"}},"official":{"repos":["secondmind-labs/trieste"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/best-arm-identification-for-stochastic-rising","slug":"best-arm-identification-for-stochastic-rising","title":"Best Arm Identification for Stochastic Rising Bandits","date":"2023-02-15","arxiv_id":"2302.07510","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/best-arm-identification-for-stochastic-rising#ran","syntology_url":"https://syntology.ai/paper/2302.07510","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.07510"}},"official":{"repos":["montenegroalessandro/bestarmidsrb"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/variational-information-pursuit-for","slug":"variational-information-pursuit-for","title":"Variational Information Pursuit for Interpretable Predictions","date":"2023-02-06","arxiv_id":"2302.02876","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/variational-information-pursuit-for#ran","syntology_url":"https://syntology.ai/paper/2302.02876","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.02876"}},"official":{"repos":["ryanchankh/VariationalInformationPursuit"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/off-policy-evaluation-for-action-dependent","slug":"off-policy-evaluation-for-action-dependent","title":"Off-Policy Evaluation for Action-Dependent Non-Stationary Environments","date":"2023-01-24","arxiv_id":"2301.10330","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/off-policy-evaluation-for-action-dependent#ran","syntology_url":"https://syntology.ai/paper/2301.10330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.10330"}},"official":{"repos":["yashchandak/activens"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-conditional-cauchy-schwarz-divergence","slug":"the-conditional-cauchy-schwarz-divergence","title":"The Conditional Cauchy-Schwarz Divergence with Applications to Time-Series Data and Sequential Decision Making","date":"2023-01-21","arxiv_id":"2301.08970","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-conditional-cauchy-schwarz-divergence#ran","syntology_url":"https://syntology.ai/paper/2301.08970","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.08970"}},"official":{"repos":["sjyucnel/conditional_cs_divergence"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/autoregressive-bandits","slug":"autoregressive-bandits","title":"Autoregressive Bandits","date":"2022-12-12","arxiv_id":"2212.06251","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autoregressive-bandits#ran","syntology_url":"https://syntology.ai/paper/2212.06251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.06251"}},"official":{"repos":["gianmarcogenalti/autoregressive-bandits"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/information-theoretic-safe-exploration-with","slug":"information-theoretic-safe-exploration-with","title":"Information-Theoretic Safe Exploration with Gaussian Processes","date":"2022-12-09","arxiv_id":"2212.04914","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":2,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":8,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/information-theoretic-safe-exploration-with#ran","syntology_url":"https://syntology.ai/paper/2212.04914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.04914"}},"official":{"repos":["boschresearch/information-theoretic-safe-exploration"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/ace-cooperative-multi-agent-q-learning-with","slug":"ace-cooperative-multi-agent-q-learning-with","title":"ACE: Cooperative Multi-agent Q-learning with Bidirectional Action-Dependency","date":"2022-11-29","arxiv_id":"2211.16068","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ace-cooperative-multi-agent-q-learning-with#ran","syntology_url":"https://syntology.ai/paper/2211.16068","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.16068"}},"official":{"repos":["opendilab/ace"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unimask-unified-inference-in-sequential","slug":"unimask-unified-inference-in-sequential","title":"UniMASK: Unified Inference in Sequential Decision Problems","date":"2022-11-20","arxiv_id":"2211.10869","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/unimask-unified-inference-in-sequential#ran","syntology_url":"https://syntology.ai/paper/2211.10869","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.10869"}},"official":{"repos":["micahcarroll/unimask"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-follow-instructions-in-text-based","slug":"learning-to-follow-instructions-in-text-based","title":"Learning to Follow Instructions in Text-Based Games","date":"2022-11-08","arxiv_id":"2211.04591","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-follow-instructions-in-text-based#ran","syntology_url":"https://syntology.ai/paper/2211.04591","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.04591"}},"official":{"repos":["mathieutuli/ltl-gata"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dungeons-and-data-a-large-scale-nethack","slug":"dungeons-and-data-a-large-scale-nethack","title":"Dungeons and Data: A Large-Scale NetHack Dataset","date":"2022-11-01","arxiv_id":"2211.00539","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dungeons-and-data-a-large-scale-nethack#ran","syntology_url":"https://syntology.ai/paper/2211.00539","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.00539"}},"official":{"repos":["facebookresearch/nle"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/markup-to-image-diffusion-models-with","slug":"markup-to-image-diffusion-models-with","title":"Markup-to-Image Diffusion Models with Scheduled Sampling","date":"2022-10-11","arxiv_id":"2210.05147","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/markup-to-image-diffusion-models-with#ran","syntology_url":"https://syntology.ai/paper/2210.05147","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05147"}},"official":{"repos":["da03/markup2im"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/non-monotonic-resource-utilization-in-the","slug":"non-monotonic-resource-utilization-in-the","title":"Non-monotonic Resource Utilization in the Bandits with Knapsacks Problem","date":"2022-09-24","arxiv_id":"2209.12013","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/non-monotonic-resource-utilization-in-the#ran","syntology_url":"https://syntology.ai/paper/2209.12013","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.12013"}},"official":{"repos":["raunakkmr/non-monotonic-resource-utilization-in-the-bandits-with-knapsacks-problem-code"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/contextual-bandits-with-large-action-spaces","slug":"contextual-bandits-with-large-action-spaces","title":"Contextual Bandits with Large Action Spaces: Made Practical","date":"2022-07-12","arxiv_id":"2207.05836","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/contextual-bandits-with-large-action-spaces#ran","syntology_url":"https://syntology.ai/paper/2207.05836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05836"}},"official":{"repos":["pmineiro/linrepcb"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/transformer-neural-processes-uncertainty","slug":"transformer-neural-processes-uncertainty","title":"Transformer Neural Processes: Uncertainty-Aware Meta Learning Via Sequence Modeling","date":"2022-07-09","arxiv_id":"2207.04179","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/transformer-neural-processes-uncertainty#ran","syntology_url":"https://syntology.ai/paper/2207.04179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.04179"}},"official":{"repos":["tung-nd/tnp-pytorch"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/interactively-learning-preference-constraints","slug":"interactively-learning-preference-constraints","title":"Interactively Learning Preference Constraints in Linear Bandits","date":"2022-06-10","arxiv_id":"2206.05255","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/interactively-learning-preference-constraints#ran","syntology_url":"https://syntology.ai/paper/2206.05255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05255"}},"official":{"repos":["lasgroup/adaptive-constraint-learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/achieving-long-term-fairness-in-sequential","slug":"achieving-long-term-fairness-in-sequential","title":"Achieving Long-Term Fairness in Sequential Decision Making","date":"2022-04-04","arxiv_id":"2204.01819","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":1,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 3 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/achieving-long-term-fairness-in-sequential#ran","syntology_url":"https://syntology.ai/paper/2204.01819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.01819"}},"official":{"repos":["yaoweihu/achieving-long-term-fairness"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-entity-1","slug":"deep-reinforcement-learning-for-entity-1","title":"Deep Reinforcement Learning for Entity Alignment","date":"2022-03-07","arxiv_id":"2203.03315","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-for-entity-1#ran","syntology_url":"https://syntology.ai/paper/2203.03315","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.03315"}},"official":{"repos":["guolingbing/rlea"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/pre-trained-language-models-for-interactive","slug":"pre-trained-language-models-for-interactive","title":"Pre-Trained Language Models for Interactive Decision-Making","date":"2022-02-03","arxiv_id":"2202.01771","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/pre-trained-language-models-for-interactive#ran","syntology_url":"https://syntology.ai/paper/2202.01771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.01771"}},"official":null}},{"url":"/paper/differentially-private-regret-minimization-in","slug":"differentially-private-regret-minimization-in","title":"Differentially Private Regret Minimization in Episodic Markov Decision Processes","date":"2021-12-20","arxiv_id":"2112.10599","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/differentially-private-regret-minimization-in#ran","syntology_url":"https://syntology.ai/paper/2112.10599","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.10599"}},"official":{"repos":["xingyuzhou989/privatetabularrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sope-spectrum-of-off-policy-estimators","slug":"sope-spectrum-of-off-policy-estimators","title":"SOPE: Spectrum of Off-Policy Estimators","date":"2021-11-06","arxiv_id":"2111.03936","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sope-spectrum-of-off-policy-estimators#ran","syntology_url":"https://syntology.ai/paper/2111.03936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.03936"}},"official":{"repos":["pearl-utexas/sope"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rlds-an-ecosystem-to-generate-share-and-use","slug":"rlds-an-ecosystem-to-generate-share-and-use","title":"RLDS: an Ecosystem to Generate, Share and Use Datasets in Reinforcement Learning","date":"2021-11-04","arxiv_id":"2111.02767","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rlds-an-ecosystem-to-generate-share-and-use#ran","syntology_url":"https://syntology.ai/paper/2111.02767","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.02767"}},"official":{"repos":["google-research/rlds"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/object-aware-regularization-for-addressing","slug":"object-aware-regularization-for-addressing","title":"Object-Aware Regularization for Addressing Causal Confusion in Imitation Learning","date":"2021-10-27","arxiv_id":"2110.14118","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/object-aware-regularization-for-addressing#ran","syntology_url":"https://syntology.ai/paper/2110.14118","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.14118"}},"official":{"repos":["alinlab/oreo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-causal-bayesian-optimization","slug":"dynamic-causal-bayesian-optimization","title":"Dynamic Causal Bayesian Optimization","date":"2021-10-26","arxiv_id":"2110.13891","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dynamic-causal-bayesian-optimization#ran","syntology_url":"https://syntology.ai/paper/2110.13891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.13891"}},"official":{"repos":["neildhir/dcbo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-contextual-bandits-without-regret","slug":"neural-contextual-bandits-without-regret","title":"Neural Contextual Bandits without Regret","date":"2021-07-07","arxiv_id":"2107.03144","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/neural-contextual-bandits-without-regret#ran","syntology_url":"https://syntology.ai/paper/2107.03144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.03144"}},"official":{"repos":["pkassraie/NNUCB"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/iq-learn-inverse-soft-q-learning-for","slug":"iq-learn-inverse-soft-q-learning-for","title":"IQ-Learn: Inverse soft-Q Learning for Imitation","date":"2021-06-23","arxiv_id":"2106.12142","repositories_listed":5,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/iq-learn-inverse-soft-q-learning-for#ran","syntology_url":"https://syntology.ai/paper/2106.12142","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.12142"}},"official":{"repos":["Div99/IQ-Learn"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/robust-reinforcement-learning-under-minimax","slug":"robust-reinforcement-learning-under-minimax","title":"Robust Reinforcement Learning Under Minimax Regret for Green Security","date":"2021-06-15","arxiv_id":"2106.08413","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-reinforcement-learning-under-minimax#ran","syntology_url":"https://syntology.ai/paper/2106.08413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.08413"}},"official":{"repos":["lily-x/mirror"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/curriculum-design-for-teaching-via","slug":"curriculum-design-for-teaching-via","title":"Curriculum Design for Teaching via Demonstrations: Theory and Applications","date":"2021-06-08","arxiv_id":"2106.04696","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":1,"n_ran_checked":2,"n_instrument":6,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":9,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/curriculum-design-for-teaching-via#ran","syntology_url":"https://syntology.ai/paper/2106.04696","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04696"}},"official":{"repos":["adishs/neurips2021_curriculum-teaching-demonstrations_code"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/universal-off-policy-evaluation","slug":"universal-off-policy-evaluation","title":"Universal Off-Policy Evaluation","date":"2021-04-26","arxiv_id":"2104.12820","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/universal-off-policy-evaluation#ran","syntology_url":"https://syntology.ai/paper/2104.12820","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.12820"}},"official":{"repos":["yashchandak/UnO"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mixed-policy-gradient","slug":"mixed-policy-gradient","title":"Mixed Policy Gradient: off-policy reinforcement learning driven jointly by data and model","date":"2021-02-23","arxiv_id":"2102.11513","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mixed-policy-gradient#ran","syntology_url":"https://syntology.ai/paper/2102.11513","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.11513"}},"official":null}},{"url":"/paper/text-based-rl-agents-with-commonsense","slug":"text-based-rl-agents-with-commonsense","title":"Text-based RL Agents with Commonsense Knowledge: New Challenges, Environments and Baselines","date":"2020-10-08","arxiv_id":"2010.03790","repositories_listed":2,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/text-based-rl-agents-with-commonsense#ran","syntology_url":"https://syntology.ai/paper/2010.03790","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.03790"}},"official":{"repos":["IBM/commonsense-rl"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/occupancy-anticipation-for-efficient","slug":"occupancy-anticipation-for-efficient","title":"Occupancy Anticipation for Efficient Exploration and Navigation","date":"2020-08-21","arxiv_id":"2008.09285","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/occupancy-anticipation-for-efficient#ran","syntology_url":"https://syntology.ai/paper/2008.09285","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.09285"}},"official":{"repos":["facebookresearch/OccupancyAnticipation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-sparse-rewarded-tasks-from-sub","slug":"learning-sparse-rewarded-tasks-from-sub","title":"Learning Sparse Rewarded Tasks from Sub-Optimal Demonstrations","date":"2020-04-01","arxiv_id":"2004.00530","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-sparse-rewarded-tasks-from-sub#ran","syntology_url":"https://syntology.ai/paper/2004.00530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.00530"}},"official":null}},{"url":"/paper/learning-discrete-state-abstractions-with","slug":"learning-discrete-state-abstractions-with","title":"Learning Discrete State Abstractions With Deep Variational Inference","date":"2020-03-09","arxiv_id":"2003.04300","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-discrete-state-abstractions-with#ran","syntology_url":"https://syntology.ai/paper/2003.04300","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.04300"}},"official":{"repos":["ondrejba/discrete_abstractions"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/pddlgym-gym-environments-from-pddl-problems","slug":"pddlgym-gym-environments-from-pddl-problems","title":"PDDLGym: Gym Environments from PDDL Problems","date":"2020-02-15","arxiv_id":"2002.06432","repositories_listed":2,"syntology":{"n":11,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/pddlgym-gym-environments-from-pddl-problems#ran","syntology_url":"https://syntology.ai/paper/2002.06432","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.06432"}},"official":{"repos":["tomsilver/pddlgym"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/statistical-inference-of-the-value-function","slug":"statistical-inference-of-the-value-function","title":"Statistical Inference of the Value Function for Reinforcement Learning in Infinite Horizon Settings","date":"2020-01-13","arxiv_id":"2001.04515","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/statistical-inference-of-the-value-function#ran","syntology_url":"https://syntology.ai/paper/2001.04515","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.04515"}},"official":{"repos":["shengzhang37/SAVE"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/planning-with-goal-conditioned-policies-1","slug":"planning-with-goal-conditioned-policies-1","title":"Planning with Goal-Conditioned Policies","date":"2019-11-19","arxiv_id":"1911.08453","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/planning-with-goal-conditioned-policies-1#ran","syntology_url":"https://syntology.ai/paper/1911.08453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.08453"}},"official":null}},{"url":"/paper/thompson-sampling-via-local-uncertainty","slug":"thompson-sampling-via-local-uncertainty","title":"Thompson Sampling via Local Uncertainty","date":"2019-10-30","arxiv_id":"1910.13673","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/thompson-sampling-via-local-uncertainty#ran","syntology_url":"https://syntology.ai/paper/1910.13673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.13673"}},"official":{"repos":["Zhendong-Wang/Thompson-Sampling-via-Local-Uncertainty"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-temporal-logic","slug":"reinforcement-learning-for-temporal-logic","title":"Reinforcement Learning for Temporal Logic Control Synthesis with Probabilistic Satisfaction Guarantees","date":"2019-09-11","arxiv_id":"1909.05304","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-for-temporal-logic#ran","syntology_url":"https://syntology.ai/paper/1909.05304","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.05304"}},"official":{"repos":["grockious/lcrl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/classification-with-costly-features-as-a","slug":"classification-with-costly-features-as-a","title":"Classification with Costly Features as a Sequential Decision-Making Problem","date":"2019-09-05","arxiv_id":"1909.02564","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/classification-with-costly-features-as-a#ran","syntology_url":"https://syntology.ai/paper/1909.02564","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.02564"}},"official":{"repos":["jaromiru/cwcf"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prediction-consistency-curvature","slug":"prediction-consistency-curvature","title":"Prediction, Consistency, Curvature: Representation Learning for Locally-Linear Control","date":"2019-09-04","arxiv_id":"1909.01506","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/prediction-consistency-curvature#ran","syntology_url":"https://syntology.ai/paper/1909.01506","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.01506"}},"official":null}},{"url":"/paper/interactive-machine-comprehension-with","slug":"interactive-machine-comprehension-with","title":"Interactive Machine Comprehension with Information Seeking Agents","date":"2019-08-27","arxiv_id":"1908.10449","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/interactive-machine-comprehension-with#ran","syntology_url":"https://syntology.ai/paper/1908.10449","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.10449"}},"official":{"repos":["xingdi-eric-yuan/imrc_public"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-learning-for-efficient-reinforcement","slug":"reward-learning-for-efficient-reinforcement","title":"Reward Learning for Efficient Reinforcement Learning in Extractive Document Summarisation","date":"2019-07-30","arxiv_id":"1907.12894","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reward-learning-for-efficient-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1907.12894","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.12894"}},"official":{"repos":["UKPLab/ijcai2019-relis"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/quizbowl-the-case-for-incremental-question","slug":"quizbowl-the-case-for-incremental-question","title":"Quizbowl: The Case for Incremental Question Answering","date":"2019-04-09","arxiv_id":"1904.04792","repositories_listed":1,"syntology":{"n":14,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quizbowl-the-case-for-incremental-question#ran","syntology_url":"https://syntology.ai/paper/1904.04792","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.04792"}},"official":null}},{"url":"/paper/fairness-with-dynamics","slug":"fairness-with-dynamics","title":"Algorithms for Fairness in Sequential Decision Making","date":"2019-01-24","arxiv_id":"1901.08568","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fairness-with-dynamics#ran","syntology_url":"https://syntology.ai/paper/1901.08568","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08568"}},"official":{"repos":["wmgithub/fairness"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-imbalanced","slug":"deep-reinforcement-learning-for-imbalanced","title":"Deep Reinforcement Learning for Imbalanced Classification","date":"2019-01-05","arxiv_id":"1901.01379","repositories_listed":3,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-for-imbalanced#ran","syntology_url":"https://syntology.ai/paper/1901.01379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.01379"}},"official":{"repos":["linenus/DRL-For-imbalanced-Classification"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}}],"record_sha256":"969806cba9a2239be0c33922c5939596cc8b97528d5ed9912f7d175c7e800ee2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}