{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/21","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":21,"pages_in_order":135,"rows_per_page":100,"rows":[2001,2100],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/20","next":"/task/reinforcement-learning-2/papers/22","papers":[{"url":"/paper/distributional-constrained-reinforcement","slug":"distributional-constrained-reinforcement","title":"Distributional constrained reinforcement learning for supply chain optimization","date":"2023-02-03","arxiv_id":"2302.01727","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-optimize-for-reinforcement","slug":"learning-to-optimize-for-reinforcement","title":"Learning to Optimize for Reinforcement Learning","date":"2023-02-03","arxiv_id":"2302.01470","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-optimize-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2302.01470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.01470"}},"official":{"repos":["sail-sg/optim4rl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/two-stage-constrained-actor-critic-fo-short","slug":"two-stage-constrained-actor-critic-fo-short","title":"Two-Stage Constrained Actor-Critic for Short Video Recommendation","date":"2023-02-03","arxiv_id":"2302.01680","repositories_listed":1,"syntology":null},{"url":"/paper/policy-expansion-for-bridging-offline-to","slug":"policy-expansion-for-bridging-offline-to","title":"Policy Expansion for Bridging Offline-to-Online Reinforcement Learning","date":"2023-02-02","arxiv_id":"2302.00935","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/policy-expansion-for-bridging-offline-to#ran","syntology_url":"https://syntology.ai/paper/2302.00935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00935"}},"official":{"repos":["haichao-zhang/pex"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/internally-rewarded-reinforcement-learning","slug":"internally-rewarded-reinforcement-learning","title":"Internally Rewarded Reinforcement Learning","date":"2023-02-01","arxiv_id":"2302.00270","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/internally-rewarded-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2302.00270","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00270"}},"official":{"repos":["mengdi-li/internally-rewarded-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-reinforcement-learning-framework-for-3","slug":"a-reinforcement-learning-framework-for-3","title":"A Reinforcement Learning Framework for Dynamic Mediation Analysis","date":"2023-01-31","arxiv_id":"2301.13348","repositories_listed":1,"syntology":null},{"url":"/paper/crc-rl-a-novel-visual-feature-representation","slug":"crc-rl-a-novel-visual-feature-representation","title":"CRC-RL: A Novel Visual Feature Representation Architecture for Unsupervised Reinforcement Learning","date":"2023-01-31","arxiv_id":"2301.13473","repositories_listed":1,"syntology":null},{"url":"/paper/execution-based-code-generation-using-deep","slug":"execution-based-code-generation-using-deep","title":"Execution-based Code Generation using Deep Reinforcement Learning","date":"2023-01-31","arxiv_id":"2301.13816","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/execution-based-code-generation-using-deep#ran","syntology_url":"https://syntology.ai/paper/2301.13816","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.13816"}},"official":{"repos":["reddy-lab-code-research/PPOCoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/few-shot-image-to-semantics-translation-for","slug":"few-shot-image-to-semantics-translation-for","title":"Few-Shot Image-to-Semantics Translation for Policy Transfer in Reinforcement Learning","date":"2023-01-31","arxiv_id":"2301.13343","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-transport-perturbations-for-safe","slug":"optimal-transport-perturbations-for-safe","title":"Optimal Transport Perturbations for Safe Reinforcement Learning with Robustness Guarantees","date":"2023-01-31","arxiv_id":"2301.13375","repositories_listed":1,"syntology":null},{"url":"/paper/guiding-online-reinforcement-learning-with","slug":"guiding-online-reinforcement-learning-with","title":"Guiding Online Reinforcement Learning with Action-Free Offline Pretraining","date":"2023-01-30","arxiv_id":"2301.12876","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/guiding-online-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2301.12876","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12876"}},"official":{"repos":["vision-cair/af-guide"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/importance-weighted-actor-critic-for-optimal-1","slug":"importance-weighted-actor-critic-for-optimal-1","title":"Importance Weighted Actor-Critic for Optimal Conservative Offline Reinforcement Learning","date":"2023-01-30","arxiv_id":"2301.12714","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/importance-weighted-actor-critic-for-optimal-1#ran","syntology_url":"https://syntology.ai/paper/2301.12714","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12714"}},"official":{"repos":["zhuhl98/acrab"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/planning-multiple-epidemic-interventions-with","slug":"planning-multiple-epidemic-interventions-with","title":"Planning Multiple Epidemic Interventions with Reinforcement Learning","date":"2023-01-30","arxiv_id":"2301.12802","repositories_listed":1,"syntology":null},{"url":"/paper/apac-authorized-probability-controlled-actor","slug":"apac-authorized-probability-controlled-actor","title":"Constrained Policy Optimization with Explicit Behavior Density for Offline Reinforcement Learning","date":"2023-01-28","arxiv_id":"2301.12130","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/apac-authorized-probability-controlled-actor#ran","syntology_url":"https://syntology.ai/paper/2301.12130","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12130"}},"official":{"repos":["evalarzj/cped"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/outcome-directed-reinforcement-learning-by","slug":"outcome-directed-reinforcement-learning-by","title":"Outcome-directed Reinforcement Learning by Uncertainty & Temporal Distance-Aware Curriculum Goal Generation","date":"2023-01-27","arxiv_id":"2301.11741","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":1,"n_ran_checked":2,"n_instrument":5,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/outcome-directed-reinforcement-learning-by#ran","syntology_url":"https://syntology.ai/paper/2301.11741","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.11741"}},"official":{"repos":["jaylee0301/outpace_official"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/automatic-intrinsic-reward-shaping-for","slug":"automatic-intrinsic-reward-shaping-for","title":"Automatic Intrinsic Reward Shaping for Exploration in Deep Reinforcement Learning","date":"2023-01-26","arxiv_id":"2301.10886","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-trust-region-based-safe","slug":"efficient-trust-region-based-safe","title":"Trust Region-Based Safe Distributional Reinforcement Learning for Multiple Constraints","date":"2023-01-26","arxiv_id":"2301.10923","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-trust-region-based-safe#ran","syntology_url":"https://syntology.ai/paper/2301.10923","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.10923"}},"official":{"repos":["rllab-snu/safe-distributional-actor-critic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-from-multiple-independent-advisors","slug":"learning-from-multiple-independent-advisors","title":"Learning from Multiple Independent Advisors in Multi-agent Reinforcement Learning","date":"2023-01-26","arxiv_id":"2301.11153","repositories_listed":1,"syntology":null},{"url":"/paper/trajectory-aware-eligibility-traces-for-off","slug":"trajectory-aware-eligibility-traces-for-off","title":"Trajectory-Aware Eligibility Traces for Off-Policy Reinforcement Learning","date":"2023-01-26","arxiv_id":"2301.11321","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/trajectory-aware-eligibility-traces-for-off#ran","syntology_url":"https://syntology.ai/paper/2301.11321","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.11321"}},"official":{"repos":["brett-daley/trajectory-aware-etraces"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/which-experiences-are-influential-for-your","slug":"which-experiences-are-influential-for-your","title":"Which Experiences Are Influential for Your Agent? Policy Iteration with Turn-over Dropout","date":"2023-01-26","arxiv_id":"2301.11168","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-control-of-partial-differential","slug":"distributed-control-of-partial-differential","title":"Distributed Control of Partial Differential Equations Using Convolutional Reinforcement Learning","date":"2023-01-25","arxiv_id":"2301.10737","repositories_listed":1,"syntology":null},{"url":"/paper/select-and-trade-towards-unified-pair-trading","slug":"select-and-trade-towards-unified-pair-trading","title":"Select and Trade: Towards Unified Pair Trading with Hierarchical Reinforcement Learning","date":"2023-01-25","arxiv_id":"2301.10724","repositories_listed":1,"syntology":null},{"url":"/paper/constrained-reinforcement-learning-for-2","slug":"constrained-reinforcement-learning-for-2","title":"Constrained Reinforcement Learning for Dexterous Manipulation","date":"2023-01-24","arxiv_id":"2301.09766","repositories_listed":1,"syntology":null},{"url":"/paper/the-configurable-tree-graph-ct-graph","slug":"the-configurable-tree-graph-ct-graph","title":"The configurable tree graph (CT-graph): measurable problems in partially observable and distal reward environments for lifelong reinforcement learning","date":"2023-01-21","arxiv_id":"2302.10887","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based-path","slug":"deep-reinforcement-learning-based-path","title":"Deep-Reinforcement-Learning-based Path Planning for Industrial Robots using Distance Sensors as Observation","date":"2023-01-14","arxiv_id":"2301.05980","repositories_listed":1,"syntology":null},{"url":"/paper/mutation-testing-of-deep-reinforcement","slug":"mutation-testing-of-deep-reinforcement","title":"Mutation Testing of Deep Reinforcement Learning Based on Real Faults","date":"2023-01-13","arxiv_id":"2301.05651","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-online-multi-task-reinforcement","slug":"adversarial-online-multi-task-reinforcement","title":"Adversarial Online Multi-Task Reinforcement Learning","date":"2023-01-11","arxiv_id":"2301.04268","repositories_listed":1,"syntology":null},{"url":"/paper/hint-assisted-reinforcement-learning-an","slug":"hint-assisted-reinforcement-learning-an","title":"Hint assisted reinforcement learning: an application in radio astronomy","date":"2023-01-10","arxiv_id":"2301.03933","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-perceive-in-deep-model-free","slug":"learning-to-perceive-in-deep-model-free","title":"Learning to Perceive in Deep Model-Free Reinforcement Learning","date":"2023-01-10","arxiv_id":"2301.03730","repositories_listed":1,"syntology":null},{"url":"/paper/orbit-a-unified-simulation-framework-for","slug":"orbit-a-unified-simulation-framework-for","title":"Orbit: A Unified Simulation Framework for Interactive Robot Learning Environments","date":"2023-01-10","arxiv_id":"2301.04195","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/orbit-a-unified-simulation-framework-for#ran","syntology_url":"https://syntology.ai/paper/2301.04195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.04195"}},"official":{"repos":["NVIDIA-Omniverse/Orbit"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/schlably-a-python-framework-for-deep","slug":"schlably-a-python-framework-for-deep","title":"schlably: A Python Framework for Deep Reinforcement Learning Based Scheduling Experiments","date":"2023-01-10","arxiv_id":"2301.04182","repositories_listed":1,"syntology":null},{"url":"/paper/why-people-skip-music-on-predicting-music","slug":"why-people-skip-music-on-predicting-music","title":"Why People Skip Music? On Predicting Music Skips using Deep Reinforcement Learning","date":"2023-01-10","arxiv_id":"2301.03881","repositories_listed":1,"syntology":null},{"url":"/paper/emergent-collective-intelligence-from-massive","slug":"emergent-collective-intelligence-from-massive","title":"Emergent collective intelligence from massive-agent cooperation and competition","date":"2023-01-04","arxiv_id":"2301.01609","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/emergent-collective-intelligence-from-massive#ran","syntology_url":"https://syntology.ai/paper/2301.01609","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.01609"}},"official":{"repos":["hanmochen/lux-open"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robofriend-an-adpative-storytelling-robotic","slug":"robofriend-an-adpative-storytelling-robotic","title":"Robofriend: An Adpative Storytelling Robotic Teddy Bear - Technical Report","date":"2023-01-04","arxiv_id":"2301.01576","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-irrigation","slug":"deep-reinforcement-learning-for-irrigation","title":"Deep reinforcement learning for irrigation scheduling using high-dimensional sensor feedback","date":"2023-01-02","arxiv_id":"2301.00899","repositories_listed":1,"syntology":null},{"url":"/paper/environment-agnostic-representation-for","slug":"environment-agnostic-representation-for","title":"Environment Agnostic Representation for Visual Reinforcement Learning","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/fusing-pre-trained-language-models-with","slug":"fusing-pre-trained-language-models-with","title":"Fusing Pre-Trained Language Models With Multimodal Prompts Through Reinforcement Learning","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/goal-guided-transformer-enabled-reinforcement","slug":"goal-guided-transformer-enabled-reinforcement","title":"Goal-Guided Transformer-Enabled Reinforcement Learning for Efficient Autonomous Navigation","date":"2023-01-01","arxiv_id":"2301.00362","repositories_listed":1,"syntology":null},{"url":"/paper/self-activating-neural-ensembles-for-1","slug":"self-activating-neural-ensembles-for-1","title":"Self-Activating Neural Ensembles for Continual Reinforcement Learning","date":"2022-12-31","arxiv_id":"2301.00141","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-success-induced","slug":"reinforcement-learning-with-success-induced","title":"Reinforcement Learning with Success Induced Task Prioritization","date":"2022-12-30","arxiv_id":"2301.00691","repositories_listed":1,"syntology":null},{"url":"/paper/risk-sensitive-policy-with-distributional","slug":"risk-sensitive-policy-with-distributional","title":"Risk-Sensitive Policy with Distributional Reinforcement Learning","date":"2022-12-30","arxiv_id":"2212.14743","repositories_listed":1,"syntology":null},{"url":"/paper/transformer-in-transformer-as-backbone-for","slug":"transformer-in-transformer-as-backbone-for","title":"Transformer in Transformer as Backbone for Deep Reinforcement Learning","date":"2022-12-30","arxiv_id":"2212.14538","repositories_listed":1,"syntology":null},{"url":"/paper/lexicographic-multi-objective-reinforcement","slug":"lexicographic-multi-objective-reinforcement","title":"Lexicographic Multi-Objective Reinforcement Learning","date":"2022-12-28","arxiv_id":"2212.13769","repositories_listed":1,"syntology":null},{"url":"/paper/on-pathologies-in-kl-regularized-1","slug":"on-pathologies-in-kl-regularized-1","title":"On Pathologies in KL-Regularized Reinforcement Learning from Expert Demonstrations","date":"2022-12-28","arxiv_id":"2212.13936","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-pathologies-in-kl-regularized-1#ran","syntology_url":"https://syntology.ai/paper/2212.13936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.13936"}},"official":{"repos":["conglu1997/nppac"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/strangeness-driven-exploration-in-multi-agent","slug":"strangeness-driven-exploration-in-multi-agent","title":"Strangeness-driven Exploration in Multi-Agent Reinforcement Learning","date":"2022-12-27","arxiv_id":"2212.13448","repositories_listed":1,"syntology":null},{"url":"/paper/learning-generalizable-representations-for-1","slug":"learning-generalizable-representations-for-1","title":"Learning Generalizable Representations for Reinforcement Learning via Adaptive Meta-learner of Behavioral Similarities","date":"2022-12-26","arxiv_id":"2212.13088","repositories_listed":1,"syntology":null},{"url":"/paper/example-guided-learning-of-stochastic-human","slug":"example-guided-learning-of-stochastic-human","title":"Example-guided learning of stochastic human driving policies using deep reinforcement learning","date":"2022-12-23","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/nars-vs-reinforcement-learning-ona-vs-q","slug":"nars-vs-reinforcement-learning-ona-vs-q","title":"NARS vs. Reinforcement learning: ONA vs. Q-Learning","date":"2022-12-23","arxiv_id":"2212.12517","repositories_listed":1,"syntology":null},{"url":"/paper/certified-policy-smoothing-for-cooperative","slug":"certified-policy-smoothing-for-cooperative","title":"Certified Policy Smoothing for Cooperative Multi-Agent Reinforcement Learning","date":"2022-12-22","arxiv_id":"2212.11746","repositories_listed":1,"syntology":null},{"url":"/paper/critic-guided-decoding-for-controlled-text","slug":"critic-guided-decoding-for-controlled-text","title":"Critic-Guided Decoding for Controlled Text Generation","date":"2022-12-21","arxiv_id":"2212.10938","repositories_listed":1,"syntology":null},{"url":"/paper/generating-multiple-length-summaries-via","slug":"generating-multiple-length-summaries-via","title":"Generating Multiple-Length Summaries via Reinforcement Learning for Unsupervised Sentence Summarization","date":"2022-12-21","arxiv_id":"2212.10843","repositories_listed":1,"syntology":null},{"url":"/paper/hyperparameters-in-contextual-rl-are-highly","slug":"hyperparameters-in-contextual-rl-are-highly","title":"Hyperparameters in Contextual RL are Highly Situational","date":"2022-12-21","arxiv_id":"2212.10876","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hyperparameters-in-contextual-rl-are-highly#ran","syntology_url":"https://syntology.ai/paper/2212.10876","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.10876"}},"official":{"repos":["automl-private/crl_hpo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-reinforcement-learning-for-the-game-of","slug":"on-reinforcement-learning-for-the-game-of","title":"On Reinforcement Learning for the Game of 2048","date":"2022-12-21","arxiv_id":"2212.11087","repositories_listed":1,"syntology":null},{"url":"/paper/comparison-of-model-free-and-model-based","slug":"comparison-of-model-free-and-model-based","title":"Comparison of Model-Free and Model-Based Learning-Informed Planning for PointGoal Navigation","date":"2022-12-17","arxiv_id":"2212.08801","repositories_listed":1,"syntology":null},{"url":"/paper/offline-reinforcement-learning-for-visual","slug":"offline-reinforcement-learning-for-visual","title":"Offline Reinforcement Learning for Visual Navigation","date":"2022-12-16","arxiv_id":"2212.08244","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-training-and-execution-multi","slug":"distributed-training-and-execution-multi","title":"Distributed-Training-and-Execution Multi-Agent Reinforcement Learning for Power Control in HetNet","date":"2022-12-15","arxiv_id":"2212.07967","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-multi-agent-deep-reinforcement","slug":"hybrid-multi-agent-deep-reinforcement","title":"Hybrid Multi-agent Deep Reinforcement Learning for Autonomous Mobility on Demand Systems","date":"2022-12-14","arxiv_id":"2212.07313","repositories_listed":1,"syntology":null},{"url":"/paper/robust-policy-optimization-in-deep","slug":"robust-policy-optimization-in-deep","title":"Robust Policy Optimization in Deep Reinforcement Learning","date":"2022-12-14","arxiv_id":"2212.07536","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-policy-optimization-in-deep#ran","syntology_url":"https://syntology.ai/paper/2212.07536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.07536"}},"official":{"repos":["vwxyzjn/cleanrl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/smacv2-an-improved-benchmark-for-cooperative-1","slug":"smacv2-an-improved-benchmark-for-cooperative-1","title":"SMACv2: An Improved Benchmark for Cooperative Multi-Agent Reinforcement Learning","date":"2022-12-14","arxiv_id":"2212.07489","repositories_listed":1,"syntology":null},{"url":"/paper/modem-accelerating-visual-model-based","slug":"modem-accelerating-visual-model-based","title":"MoDem: Accelerating Visual Model-Based Reinforcement Learning with Demonstrations","date":"2022-12-12","arxiv_id":"2212.05698","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/modem-accelerating-visual-model-based#ran","syntology_url":"https://syntology.ai/paper/2212.05698","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.05698"}},"official":{"repos":["facebookresearch/modem"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-and-tree-search","slug":"reinforcement-learning-and-tree-search","title":"Reinforcement Learning and Tree Search Methods for the Unit Commitment Problem","date":"2022-12-12","arxiv_id":"2212.06001","repositories_listed":1,"syntology":null},{"url":"/paper/effects-of-spectral-normalization-in-multi","slug":"effects-of-spectral-normalization-in-multi","title":"Effects of Spectral Normalization in Multi-agent Reinforcement Learning","date":"2022-12-10","arxiv_id":"2212.05331","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/effects-of-spectral-normalization-in-multi#ran","syntology_url":"https://syntology.ai/paper/2212.05331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.05331"}},"official":{"repos":["kinalmehta/epymarl_spectral"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/specifying-behavior-preference-with-tiered","slug":"specifying-behavior-preference-with-tiered","title":"Tiered Reward: Designing Rewards for Specification and Fast Learning of Desired Behavior","date":"2022-12-07","arxiv_id":"2212.03733","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-risk-aware-bidding-with-budget","slug":"adaptive-risk-aware-bidding-with-budget","title":"Adaptive Risk-Aware Bidding with Budget Constraint in Display Advertising","date":"2022-12-06","arxiv_id":"2212.12533","repositories_listed":1,"syntology":null},{"url":"/paper/solving-the-side-chain-packing-arrangement-of","slug":"solving-the-side-chain-packing-arrangement-of","title":"Reinforcement Learning for Molecular Dynamics Optimization: A Stochastic Pontryagin Maximum Principle Approach","date":"2022-12-06","arxiv_id":"2212.03320","repositories_listed":1,"syntology":null},{"url":"/paper/state-space-closure-revisiting-endless-online","slug":"state-space-closure-revisiting-endless-online","title":"State Space Closure: Revisiting Endless Online Level Generation via Reinforcement Learning","date":"2022-12-06","arxiv_id":"2212.02951","repositories_listed":1,"syntology":null},{"url":"/paper/switching-to-discriminative-image-captioning","slug":"switching-to-discriminative-image-captioning","title":"Switching to Discriminative Image Captioning by Relieving a Bottleneck of Reinforcement Learning","date":"2022-12-06","arxiv_id":"2212.03230","repositories_listed":1,"syntology":null},{"url":"/paper/what-is-the-solution-for-state-adversarial","slug":"what-is-the-solution-for-state-adversarial","title":"What is the Solution for State-Adversarial Multi-Agent Reinforcement Learning?","date":"2022-12-06","arxiv_id":"2212.02705","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/what-is-the-solution-for-state-adversarial#ran","syntology_url":"https://syntology.ai/paper/2212.02705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.02705"}},"official":{"repos":["susanbao/rmarl_code"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/physics-informed-model-based-reinforcement","slug":"physics-informed-model-based-reinforcement","title":"Physics-Informed Model-Based Reinforcement Learning","date":"2022-12-05","arxiv_id":"2212.02179","repositories_listed":1,"syntology":null},{"url":"/paper/td3-with-reverse-kl-regularizer-for-offline","slug":"td3-with-reverse-kl-regularizer-for-offline","title":"TD3 with Reverse KL Regularizer for Offline Reinforcement Learning from Mixed Datasets","date":"2022-12-05","arxiv_id":"2212.02125","repositories_listed":1,"syntology":null},{"url":"/paper/rlogist-fast-observation-strategy-on-whole","slug":"rlogist-fast-observation-strategy-on-whole","title":"RLogist: Fast Observation Strategy on Whole-slide Images with Deep Reinforcement Learning","date":"2022-12-04","arxiv_id":"2212.01737","repositories_listed":1,"syntology":null},{"url":"/paper/meshdqn-a-deep-reinforcement-learning","slug":"meshdqn-a-deep-reinforcement-learning","title":"MeshDQN: A Deep Reinforcement Learning Framework for Improving Meshes in Computational Fluid Dynamics","date":"2022-12-02","arxiv_id":"2212.01428","repositories_listed":1,"syntology":null},{"url":"/paper/stl-based-synthesis-of-feedback-controllers","slug":"stl-based-synthesis-of-feedback-controllers","title":"STL-Based Synthesis of Feedback Controllers Using Reinforcement Learning","date":"2022-12-02","arxiv_id":"2212.01022","repositories_listed":1,"syntology":null},{"url":"/paper/karolos-an-open-source-reinforcement-learning","slug":"karolos-an-open-source-reinforcement-learning","title":"Karolos: An Open-Source Reinforcement Learning Framework for Robot-Task Environments","date":"2022-12-01","arxiv_id":"2212.00906","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-reinforcement-learning-through","slug":"efficient-reinforcement-learning-through","title":"Efficient Reinforcement Learning Through Trajectory Generation","date":"2022-11-30","arxiv_id":"2211.17249","repositories_listed":1,"syntology":null},{"url":"/paper/general-policy-mapping-online-continual","slug":"general-policy-mapping-online-continual","title":"General policy mapping: online continual reinforcement learning inspired on the insect brain","date":"2022-11-30","arxiv_id":"2211.16759","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-bidding-strategy-in-display","slug":"real-time-bidding-strategy-in-display","title":"Real-time Bidding Strategy in Display Advertising: An Empirical Analysis","date":"2022-11-30","arxiv_id":"2212.02222","repositories_listed":1,"syntology":null},{"url":"/paper/reinforced-language-modeling-for-end-to-end","slug":"reinforced-language-modeling-for-end-to-end","title":"KRLS: Improving End-to-End Response Generation in Task Oriented Dialog with Reinforced Keywords Learning","date":"2022-11-30","arxiv_id":"2211.16773","repositories_listed":1,"syntology":null},{"url":"/paper/welfare-and-fairness-in-multi-objective","slug":"welfare-and-fairness-in-multi-objective","title":"Welfare and Fairness in Multi-objective Reinforcement Learning","date":"2022-11-30","arxiv_id":"2212.01382","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/welfare-and-fairness-in-multi-objective#ran","syntology_url":"https://syntology.ai/paper/2212.01382","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.01382"}},"official":{"repos":["MuhangTian/Fair-MORL-AAMAS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/behavior-estimation-from-multi-source-data","slug":"behavior-estimation-from-multi-source-data","title":"Behavior Estimation from Multi-Source Data for Offline Reinforcement Learning","date":"2022-11-29","arxiv_id":"2211.16078","repositories_listed":1,"syntology":null},{"url":"/paper/interpreting-primal-dual-algorithms-for","slug":"interpreting-primal-dual-algorithms-for","title":"Interpreting Primal-Dual Algorithms for Constrained Multiagent Reinforcement Learning","date":"2022-11-29","arxiv_id":"2211.16069","repositories_listed":1,"syntology":null},{"url":"/paper/improved-representation-of-asymmetrical","slug":"improved-representation-of-asymmetrical","title":"Improved Representation of Asymmetrical Distances with Interval Quasimetric Embeddings","date":"2022-11-28","arxiv_id":"2211.15120","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improved-representation-of-asymmetrical#ran","syntology_url":"https://syntology.ai/paper/2211.15120","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.15120"}},"official":{"repos":["quasimetric-learning/torch-quasimetric"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/quantile-constrained-reinforcement-learning-a","slug":"quantile-constrained-reinforcement-learning-a","title":"Quantile Constrained Reinforcement Learning: A Reinforcement Learning Framework Constraining Outage Probability","date":"2022-11-28","arxiv_id":"2211.15034","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/quantile-constrained-reinforcement-learning-a#ran","syntology_url":"https://syntology.ai/paper/2211.15034","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.15034"}},"official":{"repos":["wyjung0625/qcpo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/applying-deep-reinforcement-learning-to-the","slug":"applying-deep-reinforcement-learning-to-the","title":"Applying Deep Reinforcement Learning to the HP Model for Protein Structure Prediction","date":"2022-11-27","arxiv_id":"2211.14939","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/applying-deep-reinforcement-learning-to-the#ran","syntology_url":"https://syntology.ai/paper/2211.14939","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.14939"}},"official":{"repos":["compsoftmatterbiophysics-cityu-hk/applying-drl-to-hp-model-for-protein-structure-prediction"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/bear-physics-principled-building-environment","slug":"bear-physics-principled-building-environment","title":"BEAR: Physics-Principled Building Environment for Control and Reinforcement Learning","date":"2022-11-27","arxiv_id":"2211.14744","repositories_listed":1,"syntology":null},{"url":"/paper/explainable-and-safe-reinforcement-learning","slug":"explainable-and-safe-reinforcement-learning","title":"Explainable and Safe Reinforcement Learning for Autonomous Air Mobility","date":"2022-11-24","arxiv_id":"2211.13474","repositories_listed":1,"syntology":null},{"url":"/paper/actively-learning-costly-reward-functions-for","slug":"actively-learning-costly-reward-functions-for","title":"Actively Learning Costly Reward Functions for Reinforcement Learning","date":"2022-11-23","arxiv_id":"2211.13260","repositories_listed":1,"syntology":null},{"url":"/paper/masked-autoencoding-for-scalable-and","slug":"masked-autoencoding-for-scalable-and","title":"Masked Autoencoding for Scalable and Generalizable Decision Making","date":"2022-11-23","arxiv_id":"2211.12740","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/masked-autoencoding-for-scalable-and#ran","syntology_url":"https://syntology.ai/paper/2211.12740","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.12740"}},"official":{"repos":["fangchenliu/maskdp_public"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-reinforcement-learning-badminton","slug":"a-reinforcement-learning-badminton","title":"A Reinforcement Learning Badminton Environment for Simulating Player Tactics (Student Abstract)","date":"2022-11-22","arxiv_id":"2211.12234","repositories_listed":1,"syntology":null},{"url":"/paper/monte-carlo-forest-search-unsat-solver","slug":"monte-carlo-forest-search-unsat-solver","title":"UNSAT Solver Synthesis via Monte Carlo Forest Search","date":"2022-11-22","arxiv_id":"2211.12581","repositories_listed":1,"syntology":null},{"url":"/paper/a-low-latency-adaptive-coding-spiking","slug":"a-low-latency-adaptive-coding-spiking","title":"A Low Latency Adaptive Coding Spiking Framework for Deep Reinforcement Learning","date":"2022-11-21","arxiv_id":"2211.11760","repositories_listed":1,"syntology":null},{"url":"/paper/examining-policy-entropy-of-reinforcement","slug":"examining-policy-entropy-of-reinforcement","title":"Examining Policy Entropy of Reinforcement Learning Agents for Personalization Tasks","date":"2022-11-21","arxiv_id":"2211.11869","repositories_listed":1,"syntology":null},{"url":"/paper/tempera-test-time-prompting-via-reinforcement","slug":"tempera-test-time-prompting-via-reinforcement","title":"TEMPERA: Test-Time Prompting via Reinforcement Learning","date":"2022-11-21","arxiv_id":"2211.11890","repositories_listed":1,"syntology":null},{"url":"/paper/tinyqmix-distributed-access-control-for-mmtc","slug":"tinyqmix-distributed-access-control-for-mmtc","title":"TinyQMIX: Distributed Access Control for mMTC via Multi-agent Reinforcement Learning","date":"2022-11-21","arxiv_id":"2211.11692","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-meta-reinforcement-learning-for","slug":"efficient-meta-reinforcement-learning-for","title":"Efficient Meta Reinforcement Learning for Preference-based Fast Adaptation","date":"2022-11-20","arxiv_id":"2211.10861","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-meta-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2211.10861","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.10861"}},"official":{"repos":["stilwell-git/adaptation-with-noisy-oracle"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-search-for-job-shop-scheduling","slug":"learning-to-search-for-job-shop-scheduling","title":"Deep Reinforcement Learning Guided Improvement Heuristic for Job Shop Scheduling","date":"2022-11-20","arxiv_id":"2211.10936","repositories_listed":1,"syntology":null},{"url":"/paper/safelight-a-reinforcement-learning-method","slug":"safelight-a-reinforcement-learning-method","title":"SafeLight: A Reinforcement Learning Method toward Collision-free Traffic Signal Control","date":"2022-11-20","arxiv_id":"2211.10871","repositories_listed":1,"syntology":null},{"url":"/paper/debiasing-meta-gradient-reinforcement","slug":"debiasing-meta-gradient-reinforcement","title":"Debiasing Meta-Gradient Reinforcement Learning by Learning the Outer Value Function","date":"2022-11-19","arxiv_id":"2211.10550","repositories_listed":1,"syntology":null},{"url":"/paper/reinform-selecting-paths-with-reinforcement","slug":"reinform-selecting-paths-with-reinforcement","title":"ReInform: Selecting paths with reinforcement learning for contextualized link prediction","date":"2022-11-19","arxiv_id":"2211.10688","repositories_listed":1,"syntology":null},{"url":"/paper/gosum-extractive-summarization-of-long","slug":"gosum-extractive-summarization-of-long","title":"GoSum: Extractive Summarization of Long Documents by Reinforcement Learning and Graph Organized discourse state","date":"2022-11-18","arxiv_id":"2211.10247","repositories_listed":1,"syntology":null}],"record_sha256":"cbc823f6eac0d63912a26e53c812eb793138a2f439e198ef7c425a28929c4ede","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}