{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/19","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":19,"pages_in_order":135,"rows_per_page":100,"rows":[1801,1900],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/18","next":"/task/reinforcement-learning-2/papers/20","papers":[{"url":"/paper/flipping-coins-to-estimate-pseudocounts-for","slug":"flipping-coins-to-estimate-pseudocounts-for","title":"Flipping Coins to Estimate Pseudocounts for Exploration in Reinforcement Learning","date":"2023-06-05","arxiv_id":"2306.03186","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/flipping-coins-to-estimate-pseudocounts-for#ran","syntology_url":"https://syntology.ai/paper/2306.03186","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03186"}},"official":{"repos":["samlobel/cfn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/risk-aware-reward-shaping-of-reinforcement","slug":"risk-aware-reward-shaping-of-reinforcement","title":"Risk-Aware Reward Shaping of Reinforcement Learning Agents for Autonomous Driving","date":"2023-06-05","arxiv_id":"2306.03220","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/risk-aware-reward-shaping-of-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2306.03220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03220"}},"official":{"repos":["zhang-zengjie/code_2023_iecon_shaping_wu"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/tackling-non-stationarity-in-reinforcement","slug":"tackling-non-stationarity-in-reinforcement","title":"Tackling Non-Stationarity in Reinforcement Learning via Causal-Origin Representation","date":"2023-06-05","arxiv_id":"2306.02747","repositories_listed":1,"syntology":null},{"url":"/paper/a-unified-framework-for-factorizing","slug":"a-unified-framework-for-factorizing","title":"A Unified Framework for Factorizing Distributional Value Functions for Multi-Agent Reinforcement Learning","date":"2023-06-04","arxiv_id":"2306.02430","repositories_listed":1,"syntology":null},{"url":"/paper/ma2cl-masked-attentive-contrastive-learning","slug":"ma2cl-masked-attentive-contrastive-learning","title":"MA2CL:Masked Attentive Contrastive Learning for Multi-Agent Reinforcement Learning","date":"2023-06-03","arxiv_id":"2306.02006","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":4,"n_ran_checked":5,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ma2cl-masked-attentive-contrastive-learning#ran","syntology_url":"https://syntology.ai/paper/2306.02006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.02006"}},"official":{"repos":["ustchlsong/ma2cl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-framework-for-1","slug":"deep-reinforcement-learning-framework-for-1","title":"Deep Reinforcement Learning Framework for Thoracic Diseases Classification via Prior Knowledge Guidance","date":"2023-06-02","arxiv_id":"2306.01232","repositories_listed":1,"syntology":null},{"url":"/paper/hyperparameters-in-reinforcement-learning-and","slug":"hyperparameters-in-reinforcement-learning-and","title":"Hyperparameters in Reinforcement Learning and How To Tune Them","date":"2023-06-02","arxiv_id":"2306.01324","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hyperparameters-in-reinforcement-learning-and#ran","syntology_url":"https://syntology.ai/paper/2306.01324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01324"}},"official":{"repos":["facebookresearch/how-to-autorl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/relu-to-the-rescue-improve-your-on-policy","slug":"relu-to-the-rescue-improve-your-on-policy","title":"ReLU to the Rescue: Improve Your On-Policy Actor-Critic with Positive Advantages","date":"2023-06-02","arxiv_id":"2306.01460","repositories_listed":1,"syntology":null},{"url":"/paper/symmetric-exploration-in-combinatorial","slug":"symmetric-exploration-in-combinatorial","title":"Symmetric Replay Training: Enhancing Sample Efficiency in Deep Reinforcement Learning for Combinatorial Optimization","date":"2023-06-02","arxiv_id":"2306.01276","repositories_listed":1,"syntology":null},{"url":"/paper/tackling-unbounded-state-spaces-in-continuing","slug":"tackling-unbounded-state-spaces-in-continuing","title":"Learning to Stabilize Online Reinforcement Learning in Unbounded State Spaces","date":"2023-06-02","arxiv_id":"2306.01896","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":1,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":2,"n_no_contract":1,"n_pointer_only":6,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 2 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tackling-unbounded-state-spaces-in-continuing#ran","syntology_url":"https://syntology.ai/paper/2306.01896","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01896"}},"official":{"repos":["badger-rl/stop"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/identifiability-and-generalizability-in","slug":"identifiability-and-generalizability-in","title":"Identifiability and Generalizability in Constrained Inverse Reinforcement Learning","date":"2023-06-01","arxiv_id":"2306.00629","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/identifiability-and-generalizability-in#ran","syntology_url":"https://syntology.ai/paper/2306.00629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00629"}},"official":{"repos":["andrschl/cirl"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-and-benchmarking-offline","slug":"improving-and-benchmarking-offline","title":"Improving and Benchmarking Offline Reinforcement Learning Algorithms","date":"2023-06-01","arxiv_id":"2306.00972","repositories_listed":1,"syntology":null},{"url":"/paper/normalization-enhances-generalization-in","slug":"normalization-enhances-generalization-in","title":"Normalization Enhances Generalization in Visual Reinforcement Learning","date":"2023-06-01","arxiv_id":"2306.00656","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/normalization-enhances-generalization-in#ran","syntology_url":"https://syntology.ai/paper/2306.00656","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00656"}},"official":{"repos":["lilucse/Normalization-Enhances-Generalization-in-Visual-Reinforcement-Learning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-offline-reinforcement-learning-with-real","slug":"safe-offline-reinforcement-learning-with-real","title":"Safe Offline Reinforcement Learning with Real-Time Budget Constraints","date":"2023-06-01","arxiv_id":"2306.00603","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-diffusion-policies-for-offline-1","slug":"efficient-diffusion-policies-for-offline-1","title":"Efficient Diffusion Policies for Offline Reinforcement Learning","date":"2023-05-31","arxiv_id":"2305.20081","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-diffusion-policies-for-offline-1#ran","syntology_url":"https://syntology.ai/paper/2305.20081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.20081"}},"official":{"repos":["sail-sg/edp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/latent-exploration-for-reinforcement-learning","slug":"latent-exploration-for-reinforcement-learning","title":"Latent Exploration for Reinforcement Learning","date":"2023-05-31","arxiv_id":"2305.20065","repositories_listed":1,"syntology":null},{"url":"/paper/offline-meta-reinforcement-learning-with-in","slug":"offline-meta-reinforcement-learning-with-in","title":"Offline Meta Reinforcement Learning with In-Distribution Online Adaptation","date":"2023-05-31","arxiv_id":"2305.19529","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/offline-meta-reinforcement-learning-with-in#ran","syntology_url":"https://syntology.ai/paper/2305.19529","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.19529"}},"official":{"repos":["nagisazj/idaq_public"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rosarl-reward-only-safe-reinforcement","slug":"rosarl-reward-only-safe-reinforcement","title":"ROSARL: Reward-Only Safe Reinforcement Learning","date":"2023-05-31","arxiv_id":"2306.00035","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rosarl-reward-only-safe-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2306.00035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00035"}},"official":{"repos":["geraudnt/rosarl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dhrl-fnmr-an-intelligent-multicast-routing","slug":"dhrl-fnmr-an-intelligent-multicast-routing","title":"DHRL-FNMR: An Intelligent Multicast Routing Approach Based on Deep Hierarchical Reinforcement Learning in SDN","date":"2023-05-30","arxiv_id":"2305.19077","repositories_listed":1,"syntology":null},{"url":"/paper/robust-reinforcement-learning-objectives-for","slug":"robust-reinforcement-learning-objectives-for","title":"Robust Reinforcement Learning Objectives for Sequential Recommender Systems","date":"2023-05-30","arxiv_id":"2305.18820","repositories_listed":1,"syntology":null},{"url":"/paper/subequivariant-graph-reinforcement-learning","slug":"subequivariant-graph-reinforcement-learning","title":"Subequivariant Graph Reinforcement Learning in 3D Environments","date":"2023-05-30","arxiv_id":"2305.18951","repositories_listed":1,"syntology":{"n":18,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":13,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/subequivariant-graph-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2305.18951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18951"}},"official":{"repos":["alpc91/sgrl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":13,"ran_from_kinds":["official"]}}},{"url":"/paper/temporally-layered-architecture-for-efficient","slug":"temporally-layered-architecture-for-efficient","title":"Optimizing Attention and Cognitive Control Costs Using Temporally-Layered Architectures","date":"2023-05-30","arxiv_id":"2305.18701","repositories_listed":1,"syntology":null},{"url":"/paper/provable-and-practical-efficient-exploration","slug":"provable-and-practical-efficient-exploration","title":"Provable and Practical: Efficient Exploration in Reinforcement Learning via Langevin Monte Carlo","date":"2023-05-29","arxiv_id":"2305.18246","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/provable-and-practical-efficient-exploration#ran","syntology_url":"https://syntology.ai/paper/2305.18246","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18246"}},"official":{"repos":["hmishfaq/lmc-lsvi"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/is-centralized-training-with-decentralized","slug":"is-centralized-training-with-decentralized","title":"Is Centralized Training with Decentralized Execution Framework Centralized Enough for MARL?","date":"2023-05-27","arxiv_id":"2305.17352","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/is-centralized-training-with-decentralized#ran","syntology_url":"https://syntology.ai/paper/2305.17352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17352"}},"official":{"repos":["zyh1999/cadp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/query-policy-misalignment-in-preference-based","slug":"query-policy-misalignment-in-preference-based","title":"Query-Policy Misalignment in Preference-Based Reinforcement Learning","date":"2023-05-27","arxiv_id":"2305.17400","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":1,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/query-policy-misalignment-in-preference-based#ran","syntology_url":"https://syntology.ai/paper/2305.17400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17400"}},"official":{"repos":["huxiao09/qpa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-hierarchical-approach-to-population","slug":"a-hierarchical-approach-to-population","title":"A Hierarchical Approach to Population Training for Human-AI Collaboration","date":"2023-05-26","arxiv_id":"2305.16708","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-pd-control-using-deep-reinforcement","slug":"adaptive-pd-control-using-deep-reinforcement","title":"Adaptive PD Control using Deep Reinforcement Learning for Local-Remote Teleoperation with Stochastic Time Delays","date":"2023-05-26","arxiv_id":"2305.16979","repositories_listed":1,"syntology":null},{"url":"/paper/communication-efficient-reinforcement","slug":"communication-efficient-reinforcement","title":"Communication-Efficient Reinforcement Learning in Swarm Robotic Networks for Maze Exploration","date":"2023-05-26","arxiv_id":"2305.17087","repositories_listed":1,"syntology":null},{"url":"/paper/physical-deep-reinforcement-learning-safety","slug":"physical-deep-reinforcement-learning-safety","title":"Physics-Regulated Deep Reinforcement Learning: Invariant Embeddings","date":"2023-05-26","arxiv_id":"2305.16614","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-reward-offline-preference-guided","slug":"beyond-reward-offline-preference-guided","title":"Beyond Reward: Offline Preference-guided Policy Optimization","date":"2023-05-25","arxiv_id":"2305.16217","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-reward-offline-preference-guided#ran","syntology_url":"https://syntology.ai/paper/2305.16217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.16217"}},"official":{"repos":["bkkgbkjb/oppo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/coherent-soft-imitation-learning","slug":"coherent-soft-imitation-learning","title":"Coherent Soft Imitation Learning","date":"2023-05-25","arxiv_id":"2305.16498","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coherent-soft-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/2305.16498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.16498"}},"official":{"repos":["google-deepmind/csil"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/counterfactual-explainer-framework-for-deep","slug":"counterfactual-explainer-framework-for-deep","title":"Counterfactual Explainer Framework for Deep Reinforcement Learning Models Using Policy Distillation","date":"2023-05-25","arxiv_id":"2305.16532","repositories_listed":1,"syntology":null},{"url":"/paper/generating-synergistic-formulaic-alpha","slug":"generating-synergistic-formulaic-alpha","title":"Generating Synergistic Formulaic Alpha Collections via Reinforcement Learning","date":"2023-05-25","arxiv_id":"2306.12964","repositories_listed":1,"syntology":null},{"url":"/paper/learning-safety-constraints-from","slug":"learning-safety-constraints-from","title":"Learning Safety Constraints from Demonstrations with Unknown Rewards","date":"2023-05-25","arxiv_id":"2305.16147","repositories_listed":1,"syntology":null},{"url":"/paper/market-making-with-deep-reinforcement","slug":"market-making-with-deep-reinforcement","title":"Market Making with Deep Reinforcement Learning from Limit Order Books","date":"2023-05-25","arxiv_id":"2305.15821","repositories_listed":1,"syntology":null},{"url":"/paper/proto-iterative-policy-regularized-offline-to","slug":"proto-iterative-policy-regularized-offline-to","title":"PROTO: Iterative Policy Regularized Offline-to-Online Reinforcement Learning","date":"2023-05-25","arxiv_id":"2305.15669","repositories_listed":1,"syntology":null},{"url":"/paper/reward-machine-guided-self-paced","slug":"reward-machine-guided-self-paced","title":"Reward-Machine-Guided, Self-Paced Reinforcement Learning","date":"2023-05-25","arxiv_id":"2305.16505","repositories_listed":1,"syntology":null},{"url":"/paper/the-benefits-of-being-distributional-small-1","slug":"the-benefits-of-being-distributional-small-1","title":"The Benefits of Being Distributional: Small-Loss Bounds for Reinforcement Learning","date":"2023-05-25","arxiv_id":"2305.15703","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/the-benefits-of-being-distributional-small-1#ran","syntology_url":"https://syntology.ai/paper/2305.15703","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15703"}},"official":{"repos":["kevinzhou497/distcb"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/improving-language-models-with-advantage","slug":"improving-language-models-with-advantage","title":"Leftover Lunch: Advantage-based Offline Reinforcement Learning for Language Models","date":"2023-05-24","arxiv_id":"2305.14718","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/improving-language-models-with-advantage#ran","syntology_url":"https://syntology.ai/paper/2305.14718","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14718"}},"official":{"repos":["abaheti95/lol-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/inference-time-policy-adapters-ipa-tailoring","slug":"inference-time-policy-adapters-ipa-tailoring","title":"Inference-Time Policy Adapters (IPA): Tailoring Extreme-Scale LMs without Fine-tuning","date":"2023-05-24","arxiv_id":"2305.15065","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":3,"n_ran_checked":5,"n_instrument":4,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"9 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/inference-time-policy-adapters-ipa-tailoring#ran","syntology_url":"https://syntology.ai/paper/2305.15065","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15065"}},"official":{"repos":["gximinglu/ipa"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/2305-14550","slug":"2305-14550","title":"When should we prefer Decision Transformers for Offline Reinforcement Learning?","date":"2023-05-23","arxiv_id":"2305.14550","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/2305-14550#ran","syntology_url":"https://syntology.ai/paper/2305.14550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14550"}},"official":{"repos":["prajjwal1/rl_paradigm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/conditional-mutual-information-for-1","slug":"conditional-mutual-information-for-1","title":"Conditional Mutual Information for Disentangled Representations in Reinforcement Learning","date":"2023-05-23","arxiv_id":"2305.14133","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conditional-mutual-information-for-1#ran","syntology_url":"https://syntology.ai/paper/2305.14133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14133"}},"official":{"repos":["uoe-agents/cmid"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/constrained-reinforcement-learning-for-4","slug":"constrained-reinforcement-learning-for-4","title":"Constrained Reinforcement Learning for Dynamic Material Handling","date":"2023-05-23","arxiv_id":"2305.13824","repositories_listed":1,"syntology":null},{"url":"/paper/guard-a-safe-reinforcement-learning-benchmark","slug":"guard-a-safe-reinforcement-learning-benchmark","title":"GUARD: A Safe Reinforcement Learning Benchmark","date":"2023-05-23","arxiv_id":"2305.13681","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/guard-a-safe-reinforcement-learning-benchmark#ran","syntology_url":"https://syntology.ai/paper/2305.13681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13681"}},"official":{"repos":["intelligent-control-lab/guard"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/xroute-environment-a-novel-reinforcement","slug":"xroute-environment-a-novel-reinforcement","title":"XRoute Environment: A Novel Reinforcement Learning Environment for Routing","date":"2023-05-23","arxiv_id":"2305.13823","repositories_listed":1,"syntology":null},{"url":"/paper/know-your-enemy-investigating-monte-carlo","slug":"know-your-enemy-investigating-monte-carlo","title":"Know your Enemy: Investigating Monte-Carlo Tree Search with Opponent Models in Pommerman","date":"2023-05-22","arxiv_id":"2305.13206","repositories_listed":1,"syntology":null},{"url":"/paper/multi-task-hierarchical-adversarial-inverse","slug":"multi-task-hierarchical-adversarial-inverse","title":"Multi-task Hierarchical Adversarial Inverse Reinforcement Learning","date":"2023-05-22","arxiv_id":"2305.12633","repositories_listed":1,"syntology":null},{"url":"/paper/policy-representation-via-diffusion","slug":"policy-representation-via-diffusion","title":"Policy Representation via Diffusion Probability Model for Reinforcement Learning","date":"2023-05-22","arxiv_id":"2305.13122","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/policy-representation-via-diffusion#ran","syntology_url":"https://syntology.ai/paper/2305.13122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13122"}},"official":{"repos":["bellmantimehut/dipo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/road-planning-for-slums-via-deep","slug":"road-planning-for-slums-via-deep","title":"Road Planning for Slums via Deep Reinforcement Learning","date":"2023-05-22","arxiv_id":"2305.13060","repositories_listed":1,"syntology":null},{"url":"/paper/testing-of-deep-reinforcement-learning-agents","slug":"testing-of-deep-reinforcement-learning-agents","title":"Testing of Deep Reinforcement Learning Agents with Surrogate Models","date":"2023-05-22","arxiv_id":"2305.12751","repositories_listed":1,"syntology":null},{"url":"/paper/bertrlfuzzer-a-bert-and-reinforcement","slug":"bertrlfuzzer-a-bert-and-reinforcement","title":"BertRLFuzzer: A BERT and Reinforcement Learning Based Fuzzer","date":"2023-05-21","arxiv_id":"2305.12534","repositories_listed":1,"syntology":null},{"url":"/paper/learning-diverse-risk-preferences-in","slug":"learning-diverse-risk-preferences-in","title":"Learning Diverse Risk Preferences in Population-based Self-play","date":"2023-05-19","arxiv_id":"2305.11476","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-diverse-risk-preferences-in#ran","syntology_url":"https://syntology.ai/paper/2305.11476","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11476"}},"official":{"repos":["jackory/rpbt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/contrastive-state-augmentations-for","slug":"contrastive-state-augmentations-for","title":"Contrastive State Augmentations for Reinforcement Learning-Based Recommender Systems","date":"2023-05-18","arxiv_id":"2305.11081","repositories_listed":1,"syntology":null},{"url":"/paper/massively-scalable-inverse-reinforcement","slug":"massively-scalable-inverse-reinforcement","title":"Massively Scalable Inverse Reinforcement Learning in Google Maps","date":"2023-05-18","arxiv_id":"2305.11290","repositories_listed":1,"syntology":null},{"url":"/paper/a-genetic-fuzzy-system-for-interpretable-and","slug":"a-genetic-fuzzy-system-for-interpretable-and","title":"A Genetic Fuzzy System for Interpretable and Parsimonious Reinforcement Learning Policies","date":"2023-05-17","arxiv_id":"2305.09922","repositories_listed":1,"syntology":null},{"url":"/paper/demonstration-free-autonomous-reinforcement","slug":"demonstration-free-autonomous-reinforcement","title":"Demonstration-free Autonomous Reinforcement Learning via Implicit and Bidirectional Curriculum","date":"2023-05-17","arxiv_id":"2305.09943","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":1,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":8,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/demonstration-free-autonomous-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2305.09943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.09943"}},"official":{"repos":["snu-larr/ibc_official"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/explainable-multi-agent-reinforcement","slug":"explainable-multi-agent-reinforcement","title":"Explainable Multi-Agent Reinforcement Learning for Temporal Queries","date":"2023-05-17","arxiv_id":"2305.10378","repositories_listed":1,"syntology":null},{"url":"/paper/pittsburgh-learning-classifier-systems-for","slug":"pittsburgh-learning-classifier-systems-for","title":"Pittsburgh Learning Classifier Systems for Explainable Reinforcement Learning: Comparing with XCS","date":"2023-05-17","arxiv_id":"2305.09945","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-on-google-research","slug":"an-empirical-study-on-google-research","title":"An Empirical Study on Google Research Football Multi-agent Scenarios","date":"2023-05-16","arxiv_id":"2305.09458","repositories_listed":1,"syntology":null},{"url":"/paper/graph-reinforcement-learning-for-network","slug":"graph-reinforcement-learning-for-network","title":"Graph Reinforcement Learning for Network Control via Bi-Level Optimization","date":"2023-05-16","arxiv_id":"2305.09129","repositories_listed":1,"syntology":null},{"url":"/paper/omnisafe-an-infrastructure-for-accelerating","slug":"omnisafe-an-infrastructure-for-accelerating","title":"OmniSafe: An Infrastructure for Accelerating Safe Reinforcement Learning Research","date":"2023-05-16","arxiv_id":"2305.09304","repositories_listed":1,"syntology":null},{"url":"/paper/ramario-experimental-approach-to-reptile","slug":"ramario-experimental-approach-to-reptile","title":"RAMario: Experimental Approach to Reptile Algorithm -- Reinforcement Learning for Mario","date":"2023-05-16","arxiv_id":"2305.09655","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based-exploration","slug":"deep-reinforcement-learning-based-exploration","title":"Deep Reinforcement Learning-based Exploration of Web Applications","date":"2023-05-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/legal-extractive-summarization-of-u-s-court","slug":"legal-extractive-summarization-of-u-s-court","title":"Legal Extractive Summarization of U.S. Court Opinions","date":"2023-05-15","arxiv_id":"2305.08428","repositories_listed":1,"syntology":null},{"url":"/paper/rl4f-generating-natural-language-feedback","slug":"rl4f-generating-natural-language-feedback","title":"RL4F: Generating Natural Language Feedback with Reinforcement Learning for Repairing Model Outputs","date":"2023-05-15","arxiv_id":"2305.08844","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rl4f-generating-natural-language-feedback#ran","syntology_url":"https://syntology.ai/paper/2305.08844","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.08844"}},"official":{"repos":["feyzaakyurek/rl4f"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/what-matters-in-reinforcement-learning-for","slug":"what-matters-in-reinforcement-learning-for","title":"What Matters in Reinforcement Learning for Tractography","date":"2023-05-15","arxiv_id":"2305.09041","repositories_listed":1,"syntology":null},{"url":"/paper/quantile-based-deep-reinforcement-learning","slug":"quantile-based-deep-reinforcement-learning","title":"Quantile-Based Deep Reinforcement Learning using Two-Timescale Policy Gradient Algorithms","date":"2023-05-12","arxiv_id":"2305.07248","repositories_listed":1,"syntology":null},{"url":"/paper/extracting-diagnosis-pathways-from-electronic","slug":"extracting-diagnosis-pathways-from-electronic","title":"Extracting Diagnosis Pathways from Electronic Health Records Using Deep Reinforcement Learning","date":"2023-05-10","arxiv_id":"2305.06295","repositories_listed":1,"syntology":null},{"url":"/paper/towards-scalable-adaptive-learning-with-graph","slug":"towards-scalable-adaptive-learning-with-graph","title":"Towards Scalable Adaptive Learning with Graph Neural Networks and Reinforcement Learning","date":"2023-05-10","arxiv_id":"2305.06398","repositories_listed":1,"syntology":null},{"url":"/paper/flexible-job-shop-scheduling-via-dual","slug":"flexible-job-shop-scheduling-via-dual","title":"Flexible Job Shop Scheduling via Dual Attention Network Based Reinforcement Learning","date":"2023-05-09","arxiv_id":"2305.05119","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-energy-system-scheduling-using-a","slug":"optimal-energy-system-scheduling-using-a","title":"Optimal Energy System Scheduling Using A Constraint-Aware Reinforcement Learning Algorithm","date":"2023-05-09","arxiv_id":"2305.05484","repositories_listed":1,"syntology":null},{"url":"/paper/smaclite-a-lightweight-environment-for-multi","slug":"smaclite-a-lightweight-environment-for-multi","title":"SMAClite: A Lightweight Environment for Multi-Agent Reinforcement Learning","date":"2023-05-09","arxiv_id":"2305.05566","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-reinforcement-learning-for-1","slug":"efficient-reinforcement-learning-for-1","title":"Efficient Reinforcement Learning for Autonomous Driving with Parameterized Skills and Priors","date":"2023-05-08","arxiv_id":"2305.04412","repositories_listed":1,"syntology":null},{"url":"/paper/information-design-in-multi-agent","slug":"information-design-in-multi-agent","title":"Information Design in Multi-Agent Reinforcement Learning","date":"2023-05-08","arxiv_id":"2305.06807","repositories_listed":1,"syntology":null},{"url":"/paper/local-optimization-achieves-global-optimality","slug":"local-optimization-achieves-global-optimality","title":"Local Optimization Achieves Global Optimality in Multi-Agent Reinforcement Learning","date":"2023-05-08","arxiv_id":"2305.04819","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/local-optimization-achieves-global-optimality#ran","syntology_url":"https://syntology.ai/paper/2305.04819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.04819"}},"official":{"repos":["zhaoyl18/ratio_game"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-free-reinforcement-learning-of-semantic","slug":"model-free-reinforcement-learning-of-semantic","title":"Model-free Reinforcement Learning of Semantic Communication by Stochastic Policy Gradient","date":"2023-05-05","arxiv_id":"2305.03571","repositories_listed":1,"syntology":null},{"url":"/paper/an-asynchronous-updating-reinforcement","slug":"an-asynchronous-updating-reinforcement","title":"An Asynchronous Updating Reinforcement Learning Framework for Task-oriented Dialog System","date":"2023-05-04","arxiv_id":"2305.02718","repositories_listed":1,"syntology":null},{"url":"/paper/explainable-reinforcement-learning-via-a","slug":"explainable-reinforcement-learning-via-a","title":"Explainable Reinforcement Learning via a Causal World Model","date":"2023-05-04","arxiv_id":"2305.02749","repositories_listed":1,"syntology":null},{"url":"/paper/federated-ensemble-directed-offline","slug":"federated-ensemble-directed-offline","title":"Federated Ensemble-Directed Offline Reinforcement Learning","date":"2023-05-04","arxiv_id":"2305.03097","repositories_listed":1,"syntology":null},{"url":"/paper/imap-intrinsically-motivated-adversarial","slug":"imap-intrinsically-motivated-adversarial","title":"Toward Evaluating Robustness of Reinforcement Learning with Adversarial Policy","date":"2023-05-04","arxiv_id":"2305.02605","repositories_listed":1,"syntology":null},{"url":"/paper/simple-noisy-environment-augmentation-for","slug":"simple-noisy-environment-augmentation-for","title":"Simple Noisy Environment Augmentation for Reinforcement Learning","date":"2023-05-04","arxiv_id":"2305.02882","repositories_listed":1,"syntology":null},{"url":"/paper/towards-hierarchical-policy-learning-for","slug":"towards-hierarchical-policy-learning-for","title":"Towards Hierarchical Policy Learning for Conversational Recommendation with Hypergraph-based Reinforcement Learning","date":"2023-05-04","arxiv_id":"2305.02575","repositories_listed":1,"syntology":null},{"url":"/paper/mixed-integer-optimal-control-via","slug":"mixed-integer-optimal-control-via","title":"Mixed-Integer Optimal Control via Reinforcement Learning: A Case Study on Hybrid Electric Vehicle Energy Management","date":"2023-05-02","arxiv_id":"2305.01461","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-model-free-reinforcement-1","slug":"sample-efficient-model-free-reinforcement-1","title":"Sample Efficient Model-free Reinforcement Learning from LTL Specifications with Optimality Guarantees","date":"2023-05-02","arxiv_id":"2305.01381","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/sample-efficient-model-free-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2305.01381","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.01381"}},"official":{"repos":["shaodaqian/rl-from-ltl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/online-portfolio-management-via-deep","slug":"online-portfolio-management-via-deep","title":"Online Portfolio Management via Deep Reinforcement Learning with High-Frequency Data","date":"2023-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/catch-collaborative-feature-set-search-for","slug":"catch-collaborative-feature-set-search-for","title":"Catch: Collaborative Feature Set Search for Automated Feature Engineering","date":"2023-04-30","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-achievement-structure-for-structured","slug":"learning-achievement-structure-for-structured","title":"Learning Achievement Structure for Structured Exploration in Domains with Sparse Reward","date":"2023-04-30","arxiv_id":"2305.00508","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":2,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-achievement-structure-for-structured#ran","syntology_url":"https://syntology.ai/paper/2305.00508","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.00508"}},"official":{"repos":["pairlab/iclr-23-sea"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/posterior-sampling-for-deep-reinforcement","slug":"posterior-sampling-for-deep-reinforcement","title":"Posterior Sampling for Deep Reinforcement Learning","date":"2023-04-30","arxiv_id":"2305.00477","repositories_listed":1,"syntology":null},{"url":"/paper/semi-infinitely-constrained-markov-decision","slug":"semi-infinitely-constrained-markov-decision","title":"Semi-Infinitely Constrained Markov Decision Processes and Efficient Reinforcement Learning","date":"2023-04-29","arxiv_id":"2305.00254","repositories_listed":1,"syntology":null},{"url":"/paper/x-rlflow-graph-reinforcement-learning-for","slug":"x-rlflow-graph-reinforcement-learning-for","title":"X-RLflow: Graph Reinforcement Learning for Neural Network Subgraphs Transformation","date":"2023-04-28","arxiv_id":"2304.14698","repositories_listed":1,"syntology":null},{"url":"/paper/socnavgym-a-reinforcement-learning-gym-for","slug":"socnavgym-a-reinforcement-learning-gym-for","title":"SocNavGym: A Reinforcement Learning Gym for Social Navigation","date":"2023-04-27","arxiv_id":"2304.14102","repositories_listed":1,"syntology":null},{"url":"/paper/crop-towards-distributional-shift-robust","slug":"crop-towards-distributional-shift-robust","title":"CROP: Towards Distributional-Shift Robust Reinforcement Learning using Compact Reshaped Observation Processing","date":"2023-04-26","arxiv_id":"2304.13616","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-energy-efficiency-in-metro-systems","slug":"optimizing-energy-efficiency-in-metro-systems","title":"Optimizing Energy Efficiency in Metro Systems Under Uncertainty Disturbances Using Reinforcement Learning","date":"2023-04-26","arxiv_id":"2304.13443","repositories_listed":1,"syntology":null},{"url":"/paper/quantum-natural-policy-gradients-towards","slug":"quantum-natural-policy-gradients-towards","title":"Quantum Natural Policy Gradients: Towards Sample-Efficient Reinforcement Learning","date":"2023-04-26","arxiv_id":"2304.13571","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quantum-natural-policy-gradients-towards#ran","syntology_url":"https://syntology.ai/paper/2304.13571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.13571"}},"official":{"repos":["nicomeyer96/quantum-natural-policy-gradients"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-multi-task-approach-to-robust-deep","slug":"a-multi-task-approach-to-robust-deep","title":"A Multi-Task Approach to Robust Deep Reinforcement Learning for Resource Allocation","date":"2023-04-25","arxiv_id":"2304.12660","repositories_listed":1,"syntology":null},{"url":"/paper/partially-observable-mean-field-multi-agent","slug":"partially-observable-mean-field-multi-agent","title":"Partially Observable Mean Field Multi-Agent Reinforcement Learning Based on Graph-Attention","date":"2023-04-25","arxiv_id":"2304.12653","repositories_listed":1,"syntology":null},{"url":"/paper/proto-value-networks-scaling-representation","slug":"proto-value-networks-scaling-representation","title":"Proto-Value Networks: Scaling Representation Learning with Auxiliary Tasks","date":"2023-04-25","arxiv_id":"2304.12567","repositories_listed":1,"syntology":null},{"url":"/paper/proximal-curriculum-for-reinforcement","slug":"proximal-curriculum-for-reinforcement","title":"Proximal Curriculum for Reinforcement Learning Agents","date":"2023-04-25","arxiv_id":"2304.12877","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-control-hydrodynamic-force-on-fluidic","slug":"how-to-control-hydrodynamic-force-on-fluidic","title":"How to Control Hydrodynamic Force on Fluidic Pinball via Deep Reinforcement Learning","date":"2023-04-23","arxiv_id":"2304.11526","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-approaches-for-traffic","slug":"reinforcement-learning-approaches-for-traffic","title":"Reinforcement Learning Approaches for Traffic Signal Control under Missing Data","date":"2023-04-21","arxiv_id":"2304.10722","repositories_listed":1,"syntology":null}],"record_sha256":"2da1ce431cd2dc650c19327187db4c5aaf22f959bfab37831e4f851614da0753","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}