{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/ran/1","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":4,"rows_per_page":100,"rows":[1,100],"of":398,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning/papers/ran/1","prev":null,"next":"/task/deep-reinforcement-learning/papers/ran/2","papers":[{"url":"/paper/traced-transition-aware-regret-approximation","slug":"traced-transition-aware-regret-approximation","title":"TRACED: Transition-aware Regret Approximation with Co-learnability for Environment Design","date":"2025-06-24","arxiv_id":"2506.19997","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/traced-transition-aware-regret-approximation#ran","syntology_url":"https://syntology.ai/paper/2506.19997","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.19997"}},"official":{"repos":["cho-geonwoo/traced"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-courage-to-stop-overcoming-sunk-cost","slug":"the-courage-to-stop-overcoming-sunk-cost","title":"The Courage to Stop: Overcoming Sunk Cost Fallacy in Deep Reinforcement Learning","date":"2025-06-16","arxiv_id":"2506.13672","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-courage-to-stop-overcoming-sunk-cost#ran","syntology_url":"https://syntology.ai/paper/2506.13672","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.13672"}},"official":null}},{"url":"/paper/dr-sac-distributionally-robust-soft-actor","slug":"dr-sac-distributionally-robust-soft-actor","title":"DR-SAC: Distributionally Robust Soft Actor-Critic for Reinforcement Learning under Uncertainty","date":"2025-06-14","arxiv_id":"2506.12622","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/dr-sac-distributionally-robust-soft-actor#ran","syntology_url":"https://syntology.ai/paper/2506.12622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.12622"}},"official":{"repos":["lemutisme/dr-sac"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/graph-supported-dynamic-algorithm","slug":"graph-supported-dynamic-algorithm","title":"Graph-Supported Dynamic Algorithm Configuration for Multi-Objective Combinatorial Optimization","date":"2025-05-22","arxiv_id":"2505.16471","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/graph-supported-dynamic-algorithm#ran","syntology_url":"https://syntology.ai/paper/2505.16471","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16471"}},"official":{"repos":["robbertreijnen/gs-modac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gates-cost-aware-dynamic-workflow-scheduling","slug":"gates-cost-aware-dynamic-workflow-scheduling","title":"GATES: Cost-aware Dynamic Workflow Scheduling via Graph Attention Networks and Evolution Strategy","date":"2025-05-18","arxiv_id":"2505.12355","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/gates-cost-aware-dynamic-workflow-scheduling#ran","syntology_url":"https://syntology.ai/paper/2505.12355","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12355"}},"official":{"repos":["yashen998/gates"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-cooperative-multi-agent","slug":"enhancing-cooperative-multi-agent","title":"Enhancing Cooperative Multi-Agent Reinforcement Learning with State Modelling and Adversarial Exploration","date":"2025-05-08","arxiv_id":"2505.05262","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":3,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":4,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/enhancing-cooperative-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2505.05262","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.05262"}},"official":{"repos":["ddaedalus/smpe"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/studying-the-interplay-between-the-actor-and","slug":"studying-the-interplay-between-the-actor-and","title":"Studying the Interplay Between the Actor and Critic Representations in Reinforcement Learning","date":"2025-03-08","arxiv_id":"2503.06343","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/studying-the-interplay-between-the-actor-and#ran","syntology_url":"https://syntology.ai/paper/2503.06343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06343"}},"official":{"repos":["francelico/deac-rep"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/playing-pokemon-red-via-deep-reinforcement","slug":"playing-pokemon-red-via-deep-reinforcement","title":"Playing Pokémon Red via Deep Reinforcement Learning","date":"2025-02-27","arxiv_id":"2502.19920","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/playing-pokemon-red-via-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2502.19920","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.19920"}},"official":{"repos":["MarcoMeter/neroRL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hyperspherical-normalization-for-scalable","slug":"hyperspherical-normalization-for-scalable","title":"Hyperspherical Normalization for Scalable Deep Reinforcement Learning","date":"2025-02-21","arxiv_id":"2502.15280","repositories_listed":0,"syntology":{"n":6,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hyperspherical-normalization-for-scalable#ran","syntology_url":"https://syntology.ai/paper/2502.15280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.15280"}},"official":null}},{"url":"/paper/reevaluating-policy-gradient-methods-for","slug":"reevaluating-policy-gradient-methods-for","title":"Reevaluating Policy Gradient Methods for Imperfect-Information Games","date":"2025-02-13","arxiv_id":"2502.08938","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reevaluating-policy-gradient-methods-for#ran","syntology_url":"https://syntology.ai/paper/2502.08938","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.08938"}},"official":{"repos":["gabrfarina/exp-a-spiel","nathanlct/iig-rl-benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/continual-deep-reinforcement-learning-with","slug":"continual-deep-reinforcement-learning-with","title":"Continual Deep Reinforcement Learning with Task-Agnostic Policy Distillation","date":"2024-11-25","arxiv_id":"2411.16532","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/continual-deep-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2411.16532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16532"}},"official":{"repos":["wabbajack1/tapd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-policy-gradient-methods-without-batch","slug":"deep-policy-gradient-methods-without-batch","title":"Deep Policy Gradient Methods Without Batch Updates, Target Networks, or Replay Buffers","date":"2024-11-22","arxiv_id":"2411.15370","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-policy-gradient-methods-without-batch#ran","syntology_url":"https://syntology.ai/paper/2411.15370","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.15370"}},"official":{"repos":["gauthamvasan/avg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-the-rainbow-high-performance-deep","slug":"beyond-the-rainbow-high-performance-deep","title":"Beyond The Rainbow: High Performance Deep Reinforcement Learning on a Desktop PC","date":"2024-11-06","arxiv_id":"2411.03820","repositories_listed":3,"syntology":{"n":25,"n_ran":16,"n_constructed":10,"n_ran_checked":13,"n_instrument":3,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":21,"phrase":"16 ran (of which 10 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/beyond-the-rainbow-high-performance-deep#ran","syntology_url":"https://syntology.ai/paper/2411.03820","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.03820"}},"official":{"repos":["viptankz/btr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/adopt-modified-adam-can-converge-with-any-b-2","slug":"adopt-modified-adam-can-converge-with-any-b-2","title":"ADOPT: Modified Adam Can Converge with Any $β_2$ with the Optimal Rate","date":"2024-11-05","arxiv_id":"2411.02853","repositories_listed":2,"syntology":{"n":23,"n_ran":14,"n_constructed":0,"n_ran_checked":11,"n_instrument":3,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":4,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/adopt-modified-adam-can-converge-with-any-b-2#ran","syntology_url":"https://syntology.ai/paper/2411.02853","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02853"}},"official":{"repos":["ishohei220/adopt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/learning-successor-features-the-simple-way","slug":"learning-successor-features-the-simple-way","title":"Learning Successor Features the Simple Way","date":"2024-10-29","arxiv_id":"2410.22133","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-successor-features-the-simple-way#ran","syntology_url":"https://syntology.ai/paper/2410.22133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22133"}},"official":{"repos":["raymondchua/simple_successor_features"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/streaming-deep-reinforcement-learning-finally","slug":"streaming-deep-reinforcement-learning-finally","title":"Streaming Deep Reinforcement Learning Finally Works","date":"2024-10-18","arxiv_id":"2410.14606","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/streaming-deep-reinforcement-learning-finally#ran","syntology_url":"https://syntology.ai/paper/2410.14606","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14606"}},"official":{"repos":["mohmdelsayed/streaming-drl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/simba-simplicity-bias-for-scaling-up","slug":"simba-simplicity-bias-for-scaling-up","title":"SimBa: Simplicity Bias for Scaling Up Parameters in Deep Reinforcement Learning","date":"2024-10-13","arxiv_id":"2410.09754","repositories_listed":4,"syntology":{"n":6,"n_ran":6,"n_constructed":2,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/simba-simplicity-bias-for-scaling-up#ran","syntology_url":"https://syntology.ai/paper/2410.09754","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09754"}},"official":{"repos":["sonyresearch/simba"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/improving-generalization-on-the-procgen","slug":"improving-generalization-on-the-procgen","title":"Improving Generalization on the ProcGen Benchmark with Simple Architectural Changes and Scale","date":"2024-10-13","arxiv_id":"2410.10905","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-generalization-on-the-procgen#ran","syntology_url":"https://syntology.ai/paper/2410.10905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10905"}},"official":{"repos":["anndvision/vsop-3d"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-semantic-clustering-in-deep","slug":"exploring-semantic-clustering-in-deep","title":"Exploring Semantic Clustering in Deep Reinforcement Learning for Video Games","date":"2024-09-25","arxiv_id":"2409.17411","repositories_listed":0,"syntology":{"n":17,"n_ran":9,"n_constructed":6,"n_ran_checked":7,"n_instrument":2,"n_unverified":8,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":0,"phrase":"9 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/exploring-semantic-clustering-in-deep#ran","syntology_url":"https://syntology.ai/paper/2409.17411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.17411"}},"official":null}},{"url":"/paper/improving-deep-reinforcement-learning-by","slug":"improving-deep-reinforcement-learning-by","title":"Improving Deep Reinforcement Learning by Reducing the Chain Effect of Value and Policy Churn","date":"2024-09-07","arxiv_id":"2409.04792","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":3,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-deep-reinforcement-learning-by#ran","syntology_url":"https://syntology.ai/paper/2409.04792","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.04792"}},"official":{"repos":["bluecontra/CHAIN"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mapf-gpt-imitation-learning-for-multi-agent-1","slug":"mapf-gpt-imitation-learning-for-multi-agent-1","title":"MAPF-GPT: Imitation Learning for Multi-Agent Pathfinding at Scale","date":"2024-08-29","arxiv_id":"2409.00134","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mapf-gpt-imitation-learning-for-multi-agent-1#ran","syntology_url":"https://syntology.ai/paper/2409.00134","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.00134"}},"official":{"repos":["cognitiveaisystems/mapf-gpt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-transfer-learning-for-contextual","slug":"model-based-transfer-learning-for-contextual","title":"Model-Based Transfer Learning for Contextual Reinforcement Learning","date":"2024-08-08","arxiv_id":"2408.04498","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/model-based-transfer-learning-for-contextual#ran","syntology_url":"https://syntology.ai/paper/2408.04498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.04498"}},"official":{"repos":["jhoon-cho/mbtl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/navix-scaling-minigrid-environments-with-jax","slug":"navix-scaling-minigrid-environments-with-jax","title":"NAVIX: Scaling MiniGrid Environments with JAX","date":"2024-07-28","arxiv_id":"2407.19396","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/navix-scaling-minigrid-environments-with-jax#ran","syntology_url":"https://syntology.ai/paper/2407.19396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.19396"}},"official":{"repos":["epignatelli/navix"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-world-models-with-context-aware","slug":"efficient-world-models-with-context-aware","title":"Efficient World Models with Context-Aware Tokenization","date":"2024-06-27","arxiv_id":"2406.19320","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-world-models-with-context-aware#ran","syntology_url":"https://syntology.ai/paper/2406.19320","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19320"}},"official":{"repos":["vmicheli/delta-iris"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/breaking-the-barrier-enhanced-utility-and","slug":"breaking-the-barrier-enhanced-utility-and","title":"Breaking the Barrier: Enhanced Utility and Robustness in Smoothed DRL Agents","date":"2024-06-26","arxiv_id":"2406.18062","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":3,"n_honours":5,"n_violates":1,"n_no_contract":0,"n_pointer_only":12,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 5 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/breaking-the-barrier-enhanced-utility-and#ran","syntology_url":"https://syntology.ai/paper/2406.18062","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18062"}},"official":{"repos":["trustworthy-ml-lab/robust_highutil_smoothed_drl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/failures-are-fated-but-can-be-faded","slug":"failures-are-fated-but-can-be-faded","title":"Failures Are Fated, But Can Be Faded: Characterizing and Mitigating Unwanted Behaviors in Large-Scale Vision and Language Models","date":"2024-06-11","arxiv_id":"2406.07145","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/failures-are-fated-but-can-be-faded#ran","syntology_url":"https://syntology.ai/paper/2406.07145","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07145"}},"official":{"repos":["somsagar07/FailureShiftRL"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/probabilistic-perspectives-on-error","slug":"probabilistic-perspectives-on-error","title":"Probabilistic Perspectives on Error Minimization in Adversarial Reinforcement Learning","date":"2024-06-07","arxiv_id":"2406.04724","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/probabilistic-perspectives-on-error#ran","syntology_url":"https://syntology.ai/paper/2406.04724","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04724"}},"official":{"repos":["romanbelaire/acoe-robust-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/optimizing-automatic-differentiation-with","slug":"optimizing-automatic-differentiation-with","title":"Optimizing Automatic Differentiation with Deep Reinforcement Learning","date":"2024-06-07","arxiv_id":"2406.05027","repositories_listed":0,"syntology":{"n":29,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":17,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 17 unverified","sample_list":"/paper/optimizing-automatic-differentiation-with#ran","syntology_url":"https://syntology.ai/paper/2406.05027","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05027"}},"official":null}},{"url":"/paper/amortizing-intractable-inference-in-diffusion","slug":"amortizing-intractable-inference-in-diffusion","title":"Amortizing intractable inference in diffusion models for vision, language, and control","date":"2024-05-31","arxiv_id":"2405.20971","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/amortizing-intractable-inference-in-diffusion#ran","syntology_url":"https://syntology.ai/paper/2405.20971","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20971"}},"official":{"repos":["gfnorg/diffusion-finetuning"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/verifiably-robust-conformal-prediction","slug":"verifiably-robust-conformal-prediction","title":"Verifiably Robust Conformal Prediction","date":"2024-05-29","arxiv_id":"2405.18942","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":2,"n_ran_checked":3,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":10,"phrase":"8 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/verifiably-robust-conformal-prediction#ran","syntology_url":"https://syntology.ai/paper/2405.18942","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.18942"}},"official":{"repos":["ddv-lab/Verifiably_Robust_CP"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-study-of-plasticity-loss-in-on-policy-deep","slug":"a-study-of-plasticity-loss-in-on-policy-deep","title":"A Study of Plasticity Loss in On-Policy Deep Reinforcement Learning","date":"2024-05-29","arxiv_id":"2405.19153","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/a-study-of-plasticity-loss-in-on-policy-deep#ran","syntology_url":"https://syntology.ai/paper/2405.19153","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19153"}},"official":{"repos":["awjuliani/deep-rl-plasticity"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-q-network-on-the-fly-target","slug":"adaptive-q-network-on-the-fly-target","title":"Adaptive $Q$-Network: On-the-fly Target Selection for Deep Reinforcement Learning","date":"2024-05-25","arxiv_id":"2405.16195","repositories_listed":0,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/adaptive-q-network-on-the-fly-target#ran","syntology_url":"https://syntology.ai/paper/2405.16195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16195"}},"official":null}},{"url":"/paper/robust-deep-reinforcement-learning-with-1","slug":"robust-deep-reinforcement-learning-with-1","title":"Robust Deep Reinforcement Learning with Adaptive Adversarial Perturbations in Action Space","date":"2024-05-20","arxiv_id":"2405.11982","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-deep-reinforcement-learning-with-1#ran","syntology_url":"https://syntology.ai/paper/2405.11982","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11982"}},"official":{"repos":["lqm00/a2p-sac"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-curse-of-diversity-in-ensemble-based","slug":"the-curse-of-diversity-in-ensemble-based","title":"The Curse of Diversity in Ensemble-Based Exploration","date":"2024-05-07","arxiv_id":"2405.04342","repositories_listed":2,"syntology":{"n":16,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":13,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/the-curse-of-diversity-in-ensemble-based#ran","syntology_url":"https://syntology.ai/paper/2405.04342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.04342"}},"official":{"repos":["zhixuan-lin/ensemble-rl-continuous","zhixuan-lin/ensemble-rl-discrete"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":13,"ran_from_kinds":["official"]}}},{"url":"/paper/rice-breaking-through-the-training","slug":"rice-breaking-through-the-training","title":"RICE: Breaking Through the Training Bottlenecks of Reinforcement Learning with Explanation","date":"2024-05-05","arxiv_id":"2405.03064","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rice-breaking-through-the-training#ran","syntology_url":"https://syntology.ai/paper/2405.03064","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.03064"}},"official":{"repos":["chengzelei/rice"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/dpo-meets-ppo-reinforced-token-optimization","slug":"dpo-meets-ppo-reinforced-token-optimization","title":"DPO Meets PPO: Reinforced Token Optimization for RLHF","date":"2024-04-29","arxiv_id":"2404.18922","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dpo-meets-ppo-reinforced-token-optimization#ran","syntology_url":"https://syntology.ai/paper/2404.18922","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.18922"}},"official":{"repos":["zkshan2002/rto"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/quality-diversity-actor-critic-learning-high","slug":"quality-diversity-actor-critic-learning-high","title":"Quality-Diversity Actor-Critic: Learning High-Performing and Diverse Behaviors via Value and Successor Features Critics","date":"2024-03-15","arxiv_id":"2403.09930","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/quality-diversity-actor-critic-learning-high#ran","syntology_url":"https://syntology.ai/paper/2403.09930","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.09930"}},"official":{"repos":["adaptive-intelligent-robotics/qdac"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sindy-rl-interpretable-and-efficient-model","slug":"sindy-rl-interpretable-and-efficient-model","title":"SINDy-RL: Interpretable and Efficient Model-Based Reinforcement Learning","date":"2024-03-14","arxiv_id":"2403.09110","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sindy-rl-interpretable-and-efficient-model#ran","syntology_url":"https://syntology.ai/paper/2403.09110","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.09110"}},"official":{"repos":["nzolman/sindy-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/symmetry-aware-reinforcement-learning-for","slug":"symmetry-aware-reinforcement-learning-for","title":"Symmetry-aware Reinforcement Learning for Robotic Assembly under Partial Observability with a Soft Wrist","date":"2024-02-28","arxiv_id":"2402.18002","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/symmetry-aware-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2402.18002","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18002"}},"official":{"repos":["omron-sinicx/symmetry-aware-pomdp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-data-augmentation-in-deep","slug":"revisiting-data-augmentation-in-deep","title":"Revisiting Data Augmentation in Deep Reinforcement Learning","date":"2024-02-19","arxiv_id":"2402.12181","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/revisiting-data-augmentation-in-deep#ran","syntology_url":"https://syntology.ai/paper/2402.12181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12181"}},"official":{"repos":["jianshu-hu/drqv2"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-ensemble-model","slug":"reinforcement-learning-with-ensemble-model","title":"Reinforcement Learning with Ensemble Model Predictive Safety Certification","date":"2024-02-06","arxiv_id":"2402.04182","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-ensemble-model#ran","syntology_url":"https://syntology.ai/paper/2402.04182","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.04182"}},"official":{"repos":["svengronauer/x-mpsc"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/invit-a-generalizable-routing-problem-solver","slug":"invit-a-generalizable-routing-problem-solver","title":"INViT: A Generalizable Routing Problem Solver with Invariant Nested View Transformer","date":"2024-02-04","arxiv_id":"2402.02317","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":12,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/invit-a-generalizable-routing-problem-solver#ran","syntology_url":"https://syntology.ai/paper/2402.02317","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02317"}},"official":{"repos":["kasumigaoka-utaha/invit"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-optimal-adversarial-robust-q-learning","slug":"towards-optimal-adversarial-robust-q-learning","title":"Towards Optimal Adversarial Robust Q-learning with Bellman Infinity-error","date":"2024-02-03","arxiv_id":"2402.02165","repositories_listed":2,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-optimal-adversarial-robust-q-learning#ran","syntology_url":"https://syntology.ai/paper/2402.02165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02165"}},"official":{"repos":["leoranlmia/CAR-DQN"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vertical-symbolic-regression-via-deep-policy","slug":"vertical-symbolic-regression-via-deep-policy","title":"Vertical Symbolic Regression via Deep Policy Gradient","date":"2024-02-01","arxiv_id":"2402.00254","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vertical-symbolic-regression-via-deep-policy#ran","syntology_url":"https://syntology.ai/paper/2402.00254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.00254"}},"official":{"repos":["jiangnanhugo/vsr-dpg"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploration-and-anti-exploration-with","slug":"exploration-and-anti-exploration-with","title":"Exploration and Anti-Exploration with Distributional Random Network Distillation","date":"2024-01-18","arxiv_id":"2401.09750","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":1,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exploration-and-anti-exploration-with#ran","syntology_url":"https://syntology.ai/paper/2401.09750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09750"}},"official":{"repos":["yk7333/drnd"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-state-and-history-representations","slug":"bridging-state-and-history-representations","title":"Bridging State and History Representations: Understanding Self-Predictive RL","date":"2024-01-17","arxiv_id":"2401.08898","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-state-and-history-representations#ran","syntology_url":"https://syntology.ai/paper/2401.08898","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.08898"}},"official":{"repos":["twni2016/self-predictive-rl"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/interpretable-concept-bottlenecks-to-align","slug":"interpretable-concept-bottlenecks-to-align","title":"Interpretable Concept Bottlenecks to Align Reinforcement Learning Agents","date":"2024-01-11","arxiv_id":"2401.05821","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/interpretable-concept-bottlenecks-to-align#ran","syntology_url":"https://syntology.ai/paper/2401.05821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05821"}},"official":{"repos":["k4ntz/scobots"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/i-rebalance-personalized-vehicle","slug":"i-rebalance-personalized-vehicle","title":"i-Rebalance: Personalized Vehicle Repositioning for Supply Demand Balance","date":"2024-01-09","arxiv_id":"2401.04429","repositories_listed":0,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/i-rebalance-personalized-vehicle#ran","syntology_url":"https://syntology.ai/paper/2401.04429","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.04429"}},"official":null}},{"url":"/paper/spqr-controlling-q-ensemble-independence-with-1","slug":"spqr-controlling-q-ensemble-independence-with-1","title":"SPQR: Controlling Q-ensemble Independence with Spiked Random Model for Reinforcement Learning","date":"2024-01-06","arxiv_id":"2401.03137","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/spqr-controlling-q-ensemble-independence-with-1#ran","syntology_url":"https://syntology.ai/paper/2401.03137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.03137"}},"official":{"repos":["dohyeoklee/SPQR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/xuance-a-comprehensive-and-unified-deep","slug":"xuance-a-comprehensive-and-unified-deep","title":"XuanCe: A Comprehensive and Unified Deep Reinforcement Learning Library","date":"2023-12-25","arxiv_id":"2312.16248","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/xuance-a-comprehensive-and-unified-deep#ran","syntology_url":"https://syntology.ai/paper/2312.16248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.16248"}},"official":{"repos":["agi-brain/xuance"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/madi-learning-to-mask-distractions-for","slug":"madi-learning-to-mask-distractions-for","title":"MaDi: Learning to Mask Distractions for Generalization in Visual Deep Reinforcement Learning","date":"2023-12-23","arxiv_id":"2312.15339","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/madi-learning-to-mask-distractions-for#ran","syntology_url":"https://syntology.ai/paper/2312.15339","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.15339"}},"official":{"repos":["bramgrooten/mask-distractions"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/not-all-tasks-are-equally-difficult-multi","slug":"not-all-tasks-are-equally-difficult-multi","title":"Not All Tasks Are Equally Difficult: Multi-Task Deep Reinforcement Learning with Dynamic Depth Routing","date":"2023-12-22","arxiv_id":"2312.14472","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/not-all-tasks-are-equally-difficult-multi#ran","syntology_url":"https://syntology.ai/paper/2312.14472","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14472"}},"official":{"repos":["DarkDawn233/D2R_MTRL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/no-prior-mask-eliminate-redundant-action-for","slug":"no-prior-mask-eliminate-redundant-action-for","title":"No Prior Mask: Eliminate Redundant Action for Deep Reinforcement Learning","date":"2023-12-11","arxiv_id":"2312.06258","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/no-prior-mask-eliminate-redundant-action-for#ran","syntology_url":"https://syntology.ai/paper/2312.06258","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06258"}},"official":{"repos":["zhongdy15/npm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adsorbrl-deep-multi-objective-reinforcement","slug":"adsorbrl-deep-multi-objective-reinforcement","title":"AdsorbRL: Deep Multi-Objective Reinforcement Learning for Inverse Catalysts Design","date":"2023-12-04","arxiv_id":"2312.02308","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adsorbrl-deep-multi-objective-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2312.02308","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02308"}},"official":{"repos":["rlacombe/adsorbrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-curricula-in-open-ended-worlds","slug":"learning-curricula-in-open-ended-worlds","title":"Learning Curricula in Open-Ended Worlds","date":"2023-12-03","arxiv_id":"2312.03126","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-curricula-in-open-ended-worlds#ran","syntology_url":"https://syntology.ai/paper/2312.03126","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03126"}},"official":{"repos":["facebookresearch/dcd"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/utilizing-explainability-techniques-for","slug":"utilizing-explainability-techniques-for","title":"Utilizing Explainability Techniques for Reinforcement Learning Model Assurance","date":"2023-11-27","arxiv_id":"2311.15838","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/utilizing-explainability-techniques-for#ran","syntology_url":"https://syntology.ai/paper/2311.15838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.15838"}},"official":{"repos":["mitre/arlin"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/uni-o4-unifying-online-and-offline-deep","slug":"uni-o4-unifying-online-and-offline-deep","title":"Uni-O4: Unifying Online and Offline Deep Reinforcement Learning with Multi-Step On-Policy Optimization","date":"2023-11-06","arxiv_id":"2311.03351","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uni-o4-unifying-online-and-offline-deep#ran","syntology_url":"https://syntology.ai/paper/2311.03351","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.03351"}},"official":{"repos":["Lei-Kun/Uni-O4"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/weakly-coupled-deep-q-networks","slug":"weakly-coupled-deep-q-networks","title":"Weakly Coupled Deep Q-Networks","date":"2023-10-28","arxiv_id":"2310.18803","repositories_listed":0,"syntology":{"n":10,"n_ran":5,"n_constructed":2,"n_ran_checked":3,"n_instrument":2,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":10,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/weakly-coupled-deep-q-networks#ran","syntology_url":"https://syntology.ai/paper/2310.18803","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.18803"}},"official":null}},{"url":"/paper/neuro-inspired-fragmentation-and-recall-to","slug":"neuro-inspired-fragmentation-and-recall-to","title":"Neuro-Inspired Fragmentation and Recall to Overcome Catastrophic Forgetting in Curiosity","date":"2023-10-26","arxiv_id":"2310.17537","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":4,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/neuro-inspired-fragmentation-and-recall-to#ran","syntology_url":"https://syntology.ai/paper/2310.17537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17537"}},"official":{"repos":["fietelab/farcuriosity"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-navigation-training-autonomous-vehicles","slug":"safe-navigation-training-autonomous-vehicles","title":"Safe Navigation: Training Autonomous Vehicles using Deep Reinforcement Learning in CARLA","date":"2023-10-23","arxiv_id":"2311.10735","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safe-navigation-training-autonomous-vehicles#ran","syntology_url":"https://syntology.ai/paper/2311.10735","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10735"}},"official":{"repos":["tejas-deo/safe-navigation-training-autonomous-vehicles-using-deep-reinforcement-learning-in-carla"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-multi-objective-combinatorial","slug":"neural-multi-objective-combinatorial","title":"Neural Multi-Objective Combinatorial Optimization with Diversity Enhancement","date":"2023-10-22","arxiv_id":"2310.15195","repositories_listed":1,"syntology":{"n":8,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/neural-multi-objective-combinatorial#ran","syntology_url":"https://syntology.ai/paper/2310.15195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.15195"}},"official":{"repos":["bill-cjb/nhde"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/a-partially-supervised-reinforcement-learning","slug":"a-partially-supervised-reinforcement-learning","title":"A Partially Supervised Reinforcement Learning Framework for Visual Active Search","date":"2023-10-15","arxiv_id":"2310.09689","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-partially-supervised-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2310.09689","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09689"}},"official":{"repos":["anindyasarkariith/psrl_vas"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hieros-hierarchical-imagination-on-structured","slug":"hieros-hierarchical-imagination-on-structured","title":"Hieros: Hierarchical Imagination on Structured State Space Sequence World Models","date":"2023-10-08","arxiv_id":"2310.05167","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/hieros-hierarchical-imagination-on-structured#ran","syntology_url":"https://syntology.ai/paper/2310.05167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05167"}},"official":{"repos":["snagnar/hieros"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/discovering-general-reinforcement-learning-1","slug":"discovering-general-reinforcement-learning-1","title":"Discovering General Reinforcement Learning Algorithms with Adversarial Environment Design","date":"2023-10-04","arxiv_id":"2310.02782","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/discovering-general-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2310.02782","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02782"}},"official":{"repos":["EmptyJackson/groove"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pre-training-with-synthetic-data-helps","slug":"pre-training-with-synthetic-data-helps","title":"Pre-training with Synthetic Data Helps Offline Reinforcement Learning","date":"2023-10-01","arxiv_id":"2310.00771","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pre-training-with-synthetic-data-helps#ran","syntology_url":"https://syntology.ai/paper/2310.00771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00771"}},"official":{"repos":["victor-wang-902/synthetic-pretrain-rl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/memory-gym-partially-observable-challenges-to","slug":"memory-gym-partially-observable-challenges-to","title":"Memory Gym: Towards Endless Tasks to Benchmark Memory Capabilities of Agents","date":"2023-09-29","arxiv_id":"2309.17207","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/memory-gym-partially-observable-challenges-to#ran","syntology_url":"https://syntology.ai/paper/2309.17207","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17207"}},"official":{"repos":["marcometer/endless-memory-gym"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cleanba-a-reproducible-and-efficient","slug":"cleanba-a-reproducible-and-efficient","title":"Cleanba: A Reproducible and Efficient Distributed Reinforcement Learning Platform","date":"2023-09-29","arxiv_id":"2310.00036","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/cleanba-a-reproducible-and-efficient#ran","syntology_url":"https://syntology.ai/paper/2310.00036","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00036"}},"official":{"repos":["vwxyzjn/cleanba"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-optimization-in-a-noisy-neighborhood-1","slug":"policy-optimization-in-a-noisy-neighborhood-1","title":"Policy Optimization in a Noisy Neighborhood: On Return Landscapes in Continuous Control","date":"2023-09-26","arxiv_id":"2309.14597","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/policy-optimization-in-a-noisy-neighborhood-1#ran","syntology_url":"https://syntology.ai/paper/2309.14597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.14597"}},"official":{"repos":["nathanrahn/return-landscapes"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/drl-based-trajectory-tracking-for-motion","slug":"drl-based-trajectory-tracking-for-motion","title":"DRL-Based Trajectory Tracking for Motion-Related Modules in Autonomous Driving","date":"2023-08-30","arxiv_id":"2308.15991","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/drl-based-trajectory-tracking-for-motion#ran","syntology_url":"https://syntology.ai/paper/2308.15991","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.15991"}},"official":{"repos":["marmotatzju/drl-based-trajectory-tracking"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-identify-critical-states-for","slug":"learning-to-identify-critical-states-for","title":"Learning to Identify Critical States for Reinforcement Learning from Videos","date":"2023-08-15","arxiv_id":"2308.07795","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-identify-critical-states-for#ran","syntology_url":"https://syntology.ai/paper/2308.07795","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.07795"}},"official":{"repos":["ai-initiative-kaust/videorlcs"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improvable-gap-balancing-for-multi-task","slug":"improvable-gap-balancing-for-multi-task","title":"Improvable Gap Balancing for Multi-Task Learning","date":"2023-07-28","arxiv_id":"2307.15429","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improvable-gap-balancing-for-multi-task#ran","syntology_url":"https://syntology.ai/paper/2307.15429","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.15429"}},"official":{"repos":["yanqidai/igb4mtl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/emergence-of-adaptive-circadian-rhythms-in","slug":"emergence-of-adaptive-circadian-rhythms-in","title":"Emergence of Adaptive Circadian Rhythms in Deep Reinforcement Learning","date":"2023-07-22","arxiv_id":"2307.12143","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emergence-of-adaptive-circadian-rhythms-in#ran","syntology_url":"https://syntology.ai/paper/2307.12143","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12143"}},"official":{"repos":["aqeel13932/mn_project"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pid-inspired-inductive-biases-for-deep-1","slug":"pid-inspired-inductive-biases-for-deep-1","title":"PID-Inspired Inductive Biases for Deep Reinforcement Learning in Partially Observable Control Tasks","date":"2023-07-12","arxiv_id":"2307.05891","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":1,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pid-inspired-inductive-biases-for-deep-1#ran","syntology_url":"https://syntology.ai/paper/2307.05891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.05891"}},"official":null}},{"url":"/paper/learning-symbolic-rules-over-abstract-meaning","slug":"learning-symbolic-rules-over-abstract-meaning","title":"Learning Symbolic Rules over Abstract Meaning Representations for Textual Reinforcement Learning","date":"2023-07-05","arxiv_id":"2307.02689","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-symbolic-rules-over-abstract-meaning#ran","syntology_url":"https://syntology.ai/paper/2307.02689","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.02689"}},"official":{"repos":["ibm/loa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/resetting-the-optimizer-in-deep-rl-an","slug":"resetting-the-optimizer-in-deep-rl-an","title":"Resetting the Optimizer in Deep RL: An Empirical Study","date":"2023-06-30","arxiv_id":"2306.17833","repositories_listed":0,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/resetting-the-optimizer-in-deep-rl-an#ran","syntology_url":"https://syntology.ai/paper/2306.17833","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.17833"}},"official":null}},{"url":"/paper/end-to-end-reinforcement-learning-for-online","slug":"end-to-end-reinforcement-learning-for-online","title":"Learning Coverage Paths in Unknown Environments with Deep Reinforcement Learning","date":"2023-06-29","arxiv_id":"2306.16978","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/end-to-end-reinforcement-learning-for-online#ran","syntology_url":"https://syntology.ai/paper/2306.16978","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.16978"}},"official":{"repos":["arvijj/rl-cpp"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"url":"/paper/rl4co-an-extensive-reinforcement-learning-for","slug":"rl4co-an-extensive-reinforcement-learning-for","title":"RL4CO: an Extensive Reinforcement Learning for Combinatorial Optimization Benchmark","date":"2023-06-29","arxiv_id":"2306.17100","repositories_listed":3,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rl4co-an-extensive-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2306.17100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.17100"}},"official":{"repos":["ai4co/rl4co","pytorch/rl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/ocatari-object-centric-atari-2600","slug":"ocatari-object-centric-atari-2600","title":"OCAtari: Object-Centric Atari 2600 Reinforcement Learning Environments","date":"2023-06-14","arxiv_id":"2306.08649","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ocatari-object-centric-atari-2600#ran","syntology_url":"https://syntology.ai/paper/2306.08649","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08649"}},"official":{"repos":["k4ntz/oc_atari"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-sage-scale-meta-learning-scheduled","slug":"meta-sage-scale-meta-learning-scheduled","title":"Meta-SAGE: Scale Meta-Learning Scheduled Adaptation with Guided Exploration for Mitigating Scale Shift on Combinatorial Optimization","date":"2023-06-05","arxiv_id":"2306.02688","repositories_listed":1,"syntology":{"n":7,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":7,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/meta-sage-scale-meta-learning-scheduled#ran","syntology_url":"https://syntology.ai/paper/2306.02688","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.02688"}},"official":{"repos":["kaist-silab/meta-sage"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/for-sale-state-action-representation-learning-1","slug":"for-sale-state-action-representation-learning-1","title":"For SALE: State-Action Representation Learning for Deep Reinforcement Learning","date":"2023-06-04","arxiv_id":"2306.02451","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/for-sale-state-action-representation-learning-1#ran","syntology_url":"https://syntology.ai/paper/2306.02451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.02451"}},"official":{"repos":["sfujim/td7"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/hyperparameters-in-reinforcement-learning-and","slug":"hyperparameters-in-reinforcement-learning-and","title":"Hyperparameters in Reinforcement Learning and How To Tune Them","date":"2023-06-02","arxiv_id":"2306.01324","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hyperparameters-in-reinforcement-learning-and#ran","syntology_url":"https://syntology.ai/paper/2306.01324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01324"}},"official":{"repos":["facebookresearch/how-to-autorl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/discovering-object-centric-generalized-value","slug":"discovering-object-centric-generalized-value","title":"Discovering Object-Centric Generalized Value Functions From Pixels","date":"2023-04-27","arxiv_id":"2304.13892","repositories_listed":1,"syntology":{"n":18,"n_ran":14,"n_constructed":0,"n_ran_checked":10,"n_instrument":4,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/discovering-object-centric-generalized-value#ran","syntology_url":"https://syntology.ai/paper/2304.13892","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.13892"}},"official":{"repos":["somjit77/oc_gvfs"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-datasets-and-market-environments-for","slug":"dynamic-datasets-and-market-environments-for","title":"Dynamic Datasets and Market Environments for Financial Reinforcement Learning","date":"2023-04-25","arxiv_id":"2304.13174","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-datasets-and-market-environments-for#ran","syntology_url":"https://syntology.ai/paper/2304.13174","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.13174"}},"official":{"repos":["AI4Finance-Foundation/FinRL","ai4finance-foundation/finrl-meta"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-deep-reinforcement-learning","slug":"efficient-deep-reinforcement-learning","title":"Efficient Deep Reinforcement Learning Requires Regulating Overfitting","date":"2023-04-20","arxiv_id":"2304.10466","repositories_listed":0,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/efficient-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2304.10466","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.10466"}},"official":null}},{"url":"/paper/h-tsp-hierarchically-solving-the-large-scale","slug":"h-tsp-hierarchically-solving-the-large-scale","title":"H-TSP: Hierarchically Solving the Large-Scale Travelling Salesman Problem","date":"2023-04-19","arxiv_id":"2304.09395","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/h-tsp-hierarchically-solving-the-large-scale#ran","syntology_url":"https://syntology.ai/paper/2304.09395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.09395"}},"official":{"repos":["Learning4Optimization-HUST/H-TSP"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pointerformer-deep-reinforced-multi-pointer","slug":"pointerformer-deep-reinforced-multi-pointer","title":"Pointerformer: Deep Reinforced Multi-Pointer Transformer for the Traveling Salesman Problem","date":"2023-04-19","arxiv_id":"2304.09407","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pointerformer-deep-reinforced-multi-pointer#ran","syntology_url":"https://syntology.ai/paper/2304.09407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.09407"}},"official":{"repos":["Learning4Optimization-HUST/Pointerformer"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["named_in_paper"]}}},{"url":"/paper/benchmarking-actor-critic-deep-reinforcement","slug":"benchmarking-actor-critic-deep-reinforcement","title":"Benchmarking Actor-Critic Deep Reinforcement Learning Algorithms for Robotics Control with Action Constraints","date":"2023-04-18","arxiv_id":"2304.08743","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-actor-critic-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2304.08743","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.08743"}},"official":{"repos":["omron-sinicx/action-constrained-rl-benchmark"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/synthetic-experience-replay-1","slug":"synthetic-experience-replay-1","title":"Synthetic Experience Replay","date":"2023-03-12","arxiv_id":"2303.06614","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":1,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/synthetic-experience-replay-1#ran","syntology_url":"https://syntology.ai/paper/2303.06614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06614"}},"official":{"repos":["conglu1997/SynthER"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/decision-transformer-under-random-frame","slug":"decision-transformer-under-random-frame","title":"Decision Transformer under Random Frame Dropping","date":"2023-03-03","arxiv_id":"2303.03391","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":3,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/decision-transformer-under-random-frame#ran","syntology_url":"https://syntology.ai/paper/2303.03391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03391"}},"official":{"repos":["hukz18/defog"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-uncertainty-in-value-functions","slug":"model-based-uncertainty-in-value-functions","title":"Model-Based Uncertainty in Value Functions","date":"2023-02-24","arxiv_id":"2302.12526","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/model-based-uncertainty-in-value-functions#ran","syntology_url":"https://syntology.ai/paper/2302.12526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12526"}},"official":{"repos":["boschresearch/ube-mbrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/the-dormant-neuron-phenomenon-in-deep","slug":"the-dormant-neuron-phenomenon-in-deep","title":"The Dormant Neuron Phenomenon in Deep Reinforcement Learning","date":"2023-02-24","arxiv_id":"2302.12902","repositories_listed":3,"syntology":{"n":8,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/the-dormant-neuron-phenomenon-in-deep#ran","syntology_url":"https://syntology.ai/paper/2302.12902","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12902"}},"official":{"repos":["google/dopamine"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/robust-deep-reinforcement-learning-through-2","slug":"robust-deep-reinforcement-learning-through-2","title":"Regret-Based Defense in Adversarial Reinforcement Learning","date":"2023-02-14","arxiv_id":"2302.06912","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robust-deep-reinforcement-learning-through-2#ran","syntology_url":"https://syntology.ai/paper/2302.06912","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.06912"}},"official":{"repos":["romanbelaire/robust-ccer"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/automatic-noise-filtering-with-dynamic-sparse","slug":"automatic-noise-filtering-with-dynamic-sparse","title":"Automatic Noise Filtering with Dynamic Sparse Training in Deep Reinforcement Learning","date":"2023-02-13","arxiv_id":"2302.06548","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/automatic-noise-filtering-with-dynamic-sparse#ran","syntology_url":"https://syntology.ai/paper/2302.06548","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.06548"}},"official":{"repos":["bramgrooten/automatic-noise-filtering"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-dropout-a-simple-yet-effective","slug":"sample-dropout-a-simple-yet-effective","title":"Sample Dropout: A Simple yet Effective Variance Reduction Technique in Deep Policy Optimization","date":"2023-02-05","arxiv_id":"2302.02299","repositories_listed":1,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":12,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/sample-dropout-a-simple-yet-effective#ran","syntology_url":"https://syntology.ai/paper/2302.02299","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.02299"}},"official":{"repos":["linzichuan/sdpo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/execution-based-code-generation-using-deep","slug":"execution-based-code-generation-using-deep","title":"Execution-based Code Generation using Deep Reinforcement Learning","date":"2023-01-31","arxiv_id":"2301.13816","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/execution-based-code-generation-using-deep#ran","syntology_url":"https://syntology.ai/paper/2301.13816","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.13816"}},"official":{"repos":["reddy-lab-code-research/PPOCoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/extreme-q-learning-maxent-rl-without-entropy","slug":"extreme-q-learning-maxent-rl-without-entropy","title":"Extreme Q-Learning: MaxEnt RL without Entropy","date":"2023-01-05","arxiv_id":"2301.02328","repositories_listed":4,"syntology":{"n":13,"n_ran":8,"n_constructed":5,"n_ran_checked":5,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"8 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/extreme-q-learning-maxent-rl-without-entropy#ran","syntology_url":"https://syntology.ai/paper/2301.02328","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.02328"}},"official":null}},{"url":"/paper/on-pathologies-in-kl-regularized-1","slug":"on-pathologies-in-kl-regularized-1","title":"On Pathologies in KL-Regularized Reinforcement Learning from Expert Demonstrations","date":"2022-12-28","arxiv_id":"2212.13936","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-pathologies-in-kl-regularized-1#ran","syntology_url":"https://syntology.ai/paper/2212.13936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.13936"}},"official":{"repos":["conglu1997/nppac"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-policy-optimization-in-deep","slug":"robust-policy-optimization-in-deep","title":"Robust Policy Optimization in Deep Reinforcement Learning","date":"2022-12-14","arxiv_id":"2212.07536","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-policy-optimization-in-deep#ran","syntology_url":"https://syntology.ai/paper/2212.07536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.07536"}},"official":{"repos":["vwxyzjn/cleanrl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/modem-accelerating-visual-model-based","slug":"modem-accelerating-visual-model-based","title":"MoDem: Accelerating Visual Model-Based Reinforcement Learning with Demonstrations","date":"2022-12-12","arxiv_id":"2212.05698","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/modem-accelerating-visual-model-based#ran","syntology_url":"https://syntology.ai/paper/2212.05698","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.05698"}},"official":{"repos":["facebookresearch/modem"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/what-is-the-solution-for-state-adversarial","slug":"what-is-the-solution-for-state-adversarial","title":"What is the Solution for State-Adversarial Multi-Agent Reinforcement Learning?","date":"2022-12-06","arxiv_id":"2212.02705","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/what-is-the-solution-for-state-adversarial#ran","syntology_url":"https://syntology.ai/paper/2212.02705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.02705"}},"official":{"repos":["susanbao/rmarl_code"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}}],"record_sha256":"b5900732df6641bca948fd68ff04f1abe0f9cffdd1ff191c240b96206881e230","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}