{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/efficient-exploration/papers/2","list_of":"/task/efficient-exploration","task":"Efficient Exploration","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":6,"rows_per_page":100,"rows":[101,200],"of":514,"counts":{"archive_papers_tagged":514,"with_a_code_link":189,"where_syntology_ran_a_sample":57,"not_listed_spam_title":0,"listed":514,"listed_where_code_ran":57,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":48,"every_run_a_failure_of_syntologys_instrument":9,"listed_with_a_run_with_no_instrument_failure":48,"listed_every_run_a_failure_of_syntologys_instrument":9,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/efficient-exploration","prev":"/task/efficient-exploration","next":"/task/efficient-exploration/papers/3","papers":[{"url":"/paper/leco-learnable-episodic-count-for-task","slug":"leco-learnable-episodic-count-for-task","title":"LECO: Learnable Episodic Count for Task-Specific Intrinsic Reward","date":"2022-10-11","arxiv_id":"2210.05409","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":1,"n_ran_checked":1,"n_instrument":6,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":10,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/leco-learnable-episodic-count-for-task#ran","syntology_url":"https://syntology.ai/paper/2210.05409","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05409"}},"official":{"repos":["kakaobrain/leco"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-dexterous-manipulation-from-exemplar","slug":"learning-dexterous-manipulation-from-exemplar","title":"Learning Dexterous Manipulation from Exemplar Object Trajectories and Pre-Grasps","date":"2022-09-22","arxiv_id":"2209.11221","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-evaluation-of-posterior-sampling","slug":"an-empirical-evaluation-of-posterior-sampling","title":"An Empirical Evaluation of Posterior Sampling for Constrained Reinforcement Learning","date":"2022-09-08","arxiv_id":"2209.03596","repositories_listed":1,"syntology":null},{"url":"/paper/incremental-3d-scene-completion-for-safe-and","slug":"incremental-3d-scene-completion-for-safe-and","title":"SC-Explorer: Incremental 3D Scene Completion for Safe and Efficient Exploration Mapping and Planning","date":"2022-08-17","arxiv_id":"2208.08307","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/incremental-3d-scene-completion-for-safe-and#ran","syntology_url":"https://syntology.ai/paper/2208.08307","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.08307"}},"official":{"repos":["ethz-asl/ssc_exploration"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/impact-makes-a-sound-and-sound-makes-an","slug":"impact-makes-a-sound-and-sound-makes-an","title":"Impact Makes a Sound and Sound Makes an Impact: Sound Guides Representations and Explorations","date":"2022-08-04","arxiv_id":"2208.02680","repositories_listed":1,"syntology":null},{"url":"/paper/the-split-gibbs-sampler-revisited","slug":"the-split-gibbs-sampler-revisited","title":"The split Gibbs sampler revisited: improvements to its algorithmic structure and augmented target distribution","date":"2022-06-28","arxiv_id":"2206.13894","repositories_listed":1,"syntology":null},{"url":"/paper/a-langevin-like-sampler-for-discrete","slug":"a-langevin-like-sampler-for-discrete","title":"A Langevin-like Sampler for Discrete Distributions","date":"2022-06-20","arxiv_id":"2206.09914","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-langevin-like-sampler-for-discrete#ran","syntology_url":"https://syntology.ai/paper/2206.09914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.09914"}},"official":{"repos":["ruqizhang/discrete-langevin"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/static-scheduling-with-predictions-learned","slug":"static-scheduling-with-predictions-learned","title":"On Preemption and Learning in Stochastic Scheduling","date":"2022-05-31","arxiv_id":"2205.15695","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/static-scheduling-with-predictions-learned#ran","syntology_url":"https://syntology.ai/paper/2205.15695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.15695"}},"official":{"repos":["hugorichard/ml4a-scheduling"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-from-self-sampled-correct-and","slug":"learning-from-self-sampled-correct-and","title":"Learning Math Reasoning from Self-Sampled Correct and Partially-Correct Solutions","date":"2022-05-28","arxiv_id":"2205.14318","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/learning-from-self-sampled-correct-and#ran","syntology_url":"https://syntology.ai/paper/2205.14318","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14318"}},"official":{"repos":["microsoft/tracecodegen"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/generating-personalized-counterfactual","slug":"generating-personalized-counterfactual","title":"Personalized Algorithmic Recourse with Preference Elicitation","date":"2022-05-27","arxiv_id":"2205.13743","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-solve-combinatorial-graph","slug":"learning-to-solve-combinatorial-graph","title":"Learning to Solve Combinatorial Graph Partitioning Problems via Efficient Exploration","date":"2022-05-27","arxiv_id":"2205.14105","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-to-solve-combinatorial-graph#ran","syntology_url":"https://syntology.ai/paper/2205.14105","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14105"}},"official":{"repos":["tomdbar/ecord"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/sigmoidally-preconditioned-off-policy","slug":"sigmoidally-preconditioned-off-policy","title":"The Sufficiency of Off-Policyness and Soft Clipping: PPO is still Insufficient according to an Off-Policy Measure","date":"2022-05-20","arxiv_id":"2205.10047","repositories_listed":1,"syntology":null},{"url":"/paper/fire-burns-sword-cuts-commonsense-inductive","slug":"fire-burns-sword-cuts-commonsense-inductive","title":"Fire Burns, Sword Cuts: Commonsense Inductive Bias for Exploration in Text-based Games","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/on-machine-learning-driven-surrogates-for","slug":"on-machine-learning-driven-surrogates-for","title":"On Machine Learning-Driven Surrogates for Sound Transmission Loss Simulations","date":"2022-04-25","arxiv_id":"2204.12290","repositories_listed":1,"syntology":null},{"url":"/paper/a-variational-approach-to-bayesian","slug":"a-variational-approach-to-bayesian","title":"A Variational Approach to Bayesian Phylogenetic Inference","date":"2022-04-16","arxiv_id":"2204.07747","repositories_listed":1,"syntology":null},{"url":"/paper/collaborative-training-of-heterogeneous","slug":"collaborative-training-of-heterogeneous","title":"Collaborative Training of Heterogeneous Reinforcement Learning Agents in Environments with Sparse Rewards: What and When to Share?","date":"2022-02-24","arxiv_id":"2202.12174","repositories_listed":1,"syntology":null},{"url":"/paper/think-global-act-local-dual-scale-graph","slug":"think-global-act-local-dual-scale-graph","title":"Think Global, Act Local: Dual-scale Graph Transformer for Vision-and-Language Navigation","date":"2022-02-23","arxiv_id":"2202.11742","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/think-global-act-local-dual-scale-graph#ran","syntology_url":"https://syntology.ai/paper/2202.11742","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.11742"}},"official":null}},{"url":"/paper/lagrangian-manifold-monte-carlo-on-monge","slug":"lagrangian-manifold-monte-carlo-on-monge","title":"Lagrangian Manifold Monte Carlo on Monge Patches","date":"2022-02-01","arxiv_id":"2202.00755","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-act-with-affordance-aware-1","slug":"learning-to-act-with-affordance-aware-1","title":"Learning to Act with Affordance-Aware Multimodal Neural SLAM","date":"2022-01-24","arxiv_id":"2201.09862","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-act-with-affordance-aware-1#ran","syntology_url":"https://syntology.ai/paper/2201.09862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.09862"}},"official":{"repos":["amazon-research/multimodal-neuralslam"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/synthesizing-explainable-counterfactual","slug":"synthesizing-explainable-counterfactual","title":"Synthesizing explainable counterfactual policies for algorithmic recourse with program synthesis","date":"2022-01-18","arxiv_id":"2201.07135","repositories_listed":1,"syntology":null},{"url":"/paper/a-fast-and-scalable-polyatomic-frank-wolfe","slug":"a-fast-and-scalable-polyatomic-frank-wolfe","title":"A Fast and Scalable Polyatomic Frank-Wolfe Algorithm for the LASSO","date":"2021-12-06","arxiv_id":"2112.02890","repositories_listed":1,"syntology":null},{"url":"/paper/noveld-a-simple-yet-effective-exploration","slug":"noveld-a-simple-yet-effective-exploration","title":"NovelD: A Simple yet Effective Exploration Criterion","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/successor-feature-landmarks-for-long-horizon","slug":"successor-feature-landmarks-for-long-horizon","title":"Successor Feature Landmarks for Long-Horizon Goal-Conditioned Reinforcement Learning","date":"2021-11-18","arxiv_id":"2111.09858","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/successor-feature-landmarks-for-long-horizon#ran","syntology_url":"https://syntology.ai/paper/2111.09858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.09858"}},"official":null}},{"url":"/paper/discovering-and-exploiting-sparse-rewards-in","slug":"discovering-and-exploiting-sparse-rewards-in","title":"Discovering and Exploiting Sparse Rewards in a Learned Behavior Space","date":"2021-11-02","arxiv_id":"2111.01919","repositories_listed":1,"syntology":null},{"url":"/paper/heterogeneous-multi-player-multi-armed","slug":"heterogeneous-multi-player-multi-armed","title":"Heterogeneous Multi-player Multi-armed Bandits: Closing the Gap and Generalization","date":"2021-10-27","arxiv_id":"2110.14622","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/heterogeneous-multi-player-multi-armed#ran","syntology_url":"https://syntology.ai/paper/2110.14622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.14622"}},"official":{"repos":["shengroup/mpmab_beacon"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/landmark-guided-subgoal-generation-in","slug":"landmark-guided-subgoal-generation-in","title":"Landmark-Guided Subgoal Generation in Hierarchical Reinforcement Learning","date":"2021-10-26","arxiv_id":"2110.13625","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/landmark-guided-subgoal-generation-in#ran","syntology_url":"https://syntology.ai/paper/2110.13625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.13625"}},"official":{"repos":["junsu-kim97/higl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/map-induction-compositional-spatial-submap-1","slug":"map-induction-compositional-spatial-submap-1","title":"Map Induction: Compositional spatial submap learning for efficient exploration in novel environments","date":"2021-10-23","arxiv_id":"2110.12301","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-skills-for-efficient-exploration","slug":"hierarchical-skills-for-efficient-exploration","title":"Hierarchical Skills for Efficient Exploration","date":"2021-10-20","arxiv_id":"2110.10809","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/hierarchical-skills-for-efficient-exploration#ran","syntology_url":"https://syntology.ai/paper/2110.10809","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.10809"}},"official":{"repos":["facebookresearch/hsd3"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/balancing-value-underestimation-and","slug":"balancing-value-underestimation-and","title":"Balancing Value Underestimation and Overestimation with Realistic Actor-Critic","date":"2021-10-19","arxiv_id":"2110.09712","repositories_listed":1,"syntology":null},{"url":"/paper/hyperdqn-a-randomized-exploration-method-for","slug":"hyperdqn-a-randomized-exploration-method-for","title":"HyperDQN: A Randomized Exploration Method for Deep Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/exploratory-state-representation-learning","slug":"exploratory-state-representation-learning","title":"Exploratory State Representation Learning","date":"2021-09-28","arxiv_id":"2109.13596","repositories_listed":1,"syntology":null},{"url":"/paper/bootstrapped-meta-learning","slug":"bootstrapped-meta-learning","title":"Bootstrapped Meta-Learning","date":"2021-09-09","arxiv_id":"2109.04504","repositories_listed":1,"syntology":null},{"url":"/paper/a-gradient-sampling-algorithm-for-stratified","slug":"a-gradient-sampling-algorithm-for-stratified","title":"A Gradient Sampling Algorithm for Stratified Maps with Applications to Topological Data Analysis","date":"2021-09-01","arxiv_id":"2109.00530","repositories_listed":1,"syntology":null},{"url":"/paper/strategically-efficient-exploration-in","slug":"strategically-efficient-exploration-in","title":"Strategically Efficient Exploration in Competitive Multi-agent Reinforcement Learning","date":"2021-07-30","arxiv_id":"2107.14698","repositories_listed":1,"syntology":null},{"url":"/paper/made-exploration-via-maximizing-deviation","slug":"made-exploration-via-maximizing-deviation","title":"MADE: Exploration via Maximizing Deviation from Explored Regions","date":"2021-06-18","arxiv_id":"2106.10268","repositories_listed":1,"syntology":null},{"url":"/paper/principled-exploration-via-optimistic","slug":"principled-exploration-via-optimistic","title":"Principled Exploration via Optimistic Bootstrapping and Backward Induction","date":"2021-05-13","arxiv_id":"2105.06022","repositories_listed":1,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":12,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/principled-exploration-via-optimistic#ran","syntology_url":"https://syntology.ai/paper/2105.06022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.06022"}},"official":{"repos":["Baichenjia/OB2I"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/deep-bandits-show-off-simple-and-efficient","slug":"deep-bandits-show-off-simple-and-efficient","title":"Deep Bandits Show-Off: Simple and Efficient Exploration with Deep Networks","date":"2021-05-10","arxiv_id":"2105.04683","repositories_listed":1,"syntology":null},{"url":"/paper/behavior-guided-actor-critic-improving","slug":"behavior-guided-actor-critic-improving","title":"Behavior-Guided Actor-Critic: Improving Exploration via Learning Policy Behavior Representation for Deep Reinforcement Learning","date":"2021-04-09","arxiv_id":"2104.04424","repositories_listed":1,"syntology":null},{"url":"/paper/nonlinear-model-reduction-for-slow-fast","slug":"nonlinear-model-reduction-for-slow-fast","title":"Nonlinear model reduction for slow-fast stochastic systems near unknown invariant manifolds","date":"2021-04-05","arxiv_id":"2104.02120","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-optimal-selection-for-composited","slug":"efficient-optimal-selection-for-composited","title":"Efficient Optimal Selection for Composited Advertising Creatives with Tree Structure","date":"2021-03-02","arxiv_id":"2103.01453","repositories_listed":1,"syntology":null},{"url":"/paper/adversarially-guided-actor-critic-1","slug":"adversarially-guided-actor-critic-1","title":"Adversarially Guided Actor-Critic","date":"2021-02-08","arxiv_id":"2102.04376","repositories_listed":1,"syntology":null},{"url":"/paper/sparse-reward-exploration-via-novelty-search","slug":"sparse-reward-exploration-via-novelty-search","title":"Sparse Reward Exploration via Novelty Search and Emitters","date":"2021-02-05","arxiv_id":"2102.03140","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-exploration-for-model-based-1","slug":"efficient-exploration-for-model-based-1","title":"Model-based Reinforcement Learning for Continuous Control with Posterior Sampling","date":"2020-11-20","arxiv_id":"2012.09613","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/efficient-exploration-for-model-based-1#ran","syntology_url":"https://syntology.ai/paper/2012.09613","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.09613"}},"official":{"repos":["yingfan-bot/mbpsrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/corrupted-contextual-bandits-with-action","slug":"corrupted-contextual-bandits-with-action","title":"A New Bandit Setting Balancing Information from State Evolution and Corrupted Context","date":"2020-11-16","arxiv_id":"2011.07989","repositories_listed":1,"syntology":null},{"url":"/paper/amortized-variational-deep-q-network","slug":"amortized-variational-deep-q-network","title":"Amortized Variational Deep Q Network","date":"2020-11-03","arxiv_id":"2011.01706","repositories_listed":1,"syntology":null},{"url":"/paper/latent-world-models-for-intrinsically","slug":"latent-world-models-for-intrinsically","title":"Latent World Models For Intrinsically Motivated Exploration","date":"2020-10-05","arxiv_id":"2010.02302","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/latent-world-models-for-intrinsically#ran","syntology_url":"https://syntology.ai/paper/2010.02302","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.02302"}},"official":{"repos":["htdt/lwm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/novelty-search-in-representational-space-for","slug":"novelty-search-in-representational-space-for","title":"Novelty Search in Representational Space for Sample Efficient Exploration","date":"2020-09-28","arxiv_id":"2009.13579","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/novelty-search-in-representational-space-for#ran","syntology_url":"https://syntology.ai/paper/2009.13579","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.13579"}},"official":{"repos":["taodav/nsrs"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/occupancy-anticipation-for-efficient","slug":"occupancy-anticipation-for-efficient","title":"Occupancy Anticipation for Efficient Exploration and Navigation","date":"2020-08-21","arxiv_id":"2008.09285","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/occupancy-anticipation-for-efficient#ran","syntology_url":"https://syntology.ai/paper/2008.09285","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.09285"}},"official":{"repos":["facebookresearch/OccupancyAnticipation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deepdrummer-generating-drum-loops-using-deep","slug":"deepdrummer-generating-drum-loops-using-deep","title":"DeepDrummer : Generating Drum Loops using Deep Learning and a Human in the Loop","date":"2020-08-10","arxiv_id":"2008.04391","repositories_listed":1,"syntology":null},{"url":"/paper/sunrise-a-simple-unified-framework-for","slug":"sunrise-a-simple-unified-framework-for","title":"SUNRISE: A Simple Unified Framework for Ensemble Learning in Deep Reinforcement Learning","date":"2020-07-09","arxiv_id":"2007.04938","repositories_listed":1,"syntology":null},{"url":"/paper/see-hear-explore-curiosity-via-audio-visual","slug":"see-hear-explore-curiosity-via-audio-visual","title":"See, Hear, Explore: Curiosity via Audio-Visual Association","date":"2020-07-07","arxiv_id":"2007.03669","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/see-hear-explore-curiosity-via-audio-visual#ran","syntology_url":"https://syntology.ai/paper/2007.03669","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.03669"}},"official":{"repos":["vdean/audio-curiosity"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchically-organized-latent-modules-for","slug":"hierarchically-organized-latent-modules-for","title":"Hierarchically Organized Latent Modules for Exploratory Search in Morphogenetic Systems","date":"2020-07-02","arxiv_id":"2007.01195","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hierarchically-organized-latent-modules-for#ran","syntology_url":"https://syntology.ai/paper/2007.01195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.01195"}},"official":{"repos":["flowersteam/holmes"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learn-to-effectively-explore-in-context-based","slug":"learn-to-effectively-explore-in-context-based","title":"MetaCURE: Meta Reinforcement Learning with Empowerment-Driven Exploration","date":"2020-06-15","arxiv_id":"2006.08170","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learn-to-effectively-explore-in-context-based#ran","syntology_url":"https://syntology.ai/paper/2006.08170","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.08170"}},"official":{"repos":["NagisaZj/MetaCURE-Public"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diversity-actor-critic-sample-aware-entropy","slug":"diversity-actor-critic-sample-aware-entropy","title":"Diversity Actor-Critic: Sample-Aware Entropy Regularization for Sample-Efficient Exploration","date":"2020-06-02","arxiv_id":"2006.01419","repositories_listed":1,"syntology":null},{"url":"/paper/multirobot-coverage-of-modular-environments","slug":"multirobot-coverage-of-modular-environments","title":"Multirobot Coverage of Modular Environments","date":"2020-05-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/provably-efficient-exploration-for-rl-with","slug":"provably-efficient-exploration-for-rl-with","title":"Provably Efficient Exploration for Reinforcement Learning Using Unsupervised Learning","date":"2020-03-15","arxiv_id":"2003.06898","repositories_listed":1,"syntology":null},{"url":"/paper/optimistic-exploration-even-with-a-1","slug":"optimistic-exploration-even-with-a-1","title":"Optimistic Exploration even with a Pessimistic Initialisation","date":"2020-02-26","arxiv_id":"2002.12174","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimistic-exploration-even-with-a-1#ran","syntology_url":"https://syntology.ai/paper/2002.12174","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.12174"}},"official":{"repos":["oxwhirl/opiq"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/glib-exploration-via-goal-literal-babbling","slug":"glib-exploration-via-goal-literal-babbling","title":"GLIB: Efficient Exploration for Relational Model-Based Reinforcement Learning via Goal-Literal Babbling","date":"2020-01-22","arxiv_id":"2001.08299","repositories_listed":1,"syntology":null},{"url":"/paper/meta-reinforcement-learning-with-autonomous-1","slug":"meta-reinforcement-learning-with-autonomous-1","title":"Meta Reinforcement Learning with Autonomous Inference of Subtask Dependencies","date":"2020-01-01","arxiv_id":"2001.00248","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":2,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meta-reinforcement-learning-with-autonomous-1#ran","syntology_url":"https://syntology.ai/paper/2001.00248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.00248"}},"official":{"repos":["srsohn/msgi"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/parameterized-indexed-value-function-for","slug":"parameterized-indexed-value-function-for","title":"Parameterized Indexed Value Function for Efficient Exploration in Reinforcement Learning","date":"2019-12-23","arxiv_id":"1912.10577","repositories_listed":1,"syntology":null},{"url":"/paper/better-exploration-with-optimistic-actor-1","slug":"better-exploration-with-optimistic-actor-1","title":"Better Exploration with Optimistic Actor Critic","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-curiosity-for-efficient-exploration","slug":"bayesian-curiosity-for-efficient-exploration","title":"Bayesian Curiosity for Efficient Exploration in Reinforcement Learning","date":"2019-11-20","arxiv_id":"1911.08701","repositories_listed":1,"syntology":null},{"url":"/paper/exploration-via-sample-efficient-subgoal","slug":"exploration-via-sample-efficient-subgoal","title":"Dynamic Subgoal-based Exploration via Bayesian Optimization","date":"2019-10-21","arxiv_id":"1910.09143","repositories_listed":1,"syntology":null},{"url":"/paper/receding-horizon-curiosity","slug":"receding-horizon-curiosity","title":"Receding Horizon Curiosity","date":"2019-10-08","arxiv_id":"1910.03620","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-seek-autonomous-source-seeking","slug":"learning-to-seek-autonomous-source-seeking","title":"Learning to Seek: Autonomous Source Seeking with Deep Reinforcement Learning Onboard a Nano Drone Microcontroller","date":"2019-09-25","arxiv_id":"1909.11236","repositories_listed":1,"syntology":null},{"url":"/paper/neural-linear-bandits-overcoming-catastrophic","slug":"neural-linear-bandits-overcoming-catastrophic","title":"Neural Linear Bandits: Overcoming Catastrophic Forgetting through Likelihood Matching","date":"2019-09-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-driven-exploration-for-reinforcement","slug":"learning-driven-exploration-for-reinforcement","title":"Learning-Driven Exploration for Reinforcement Learning","date":"2019-06-17","arxiv_id":"1906.06890","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-exploration-via-state-marginal","slug":"efficient-exploration-via-state-marginal","title":"Efficient Exploration via State Marginal Matching","date":"2019-06-12","arxiv_id":"1906.05274","repositories_listed":1,"syntology":null},{"url":"/paper/wasserstein-reinforcement-learning","slug":"wasserstein-reinforcement-learning","title":"Learning to Score Behaviors for Guided Policy Optimization","date":"2019-06-11","arxiv_id":"1906.04349","repositories_listed":1,"syntology":null},{"url":"/paper/the-minerl-competition-on-sample-efficient","slug":"the-minerl-competition-on-sample-efficient","title":"The MineRL 2019 Competition on Sample Efficient Reinforcement Learning using Human Priors","date":"2019-04-22","arxiv_id":"1904.10079","repositories_listed":1,"syntology":null},{"url":"/paper/concurrent-meta-reinforcement-learning","slug":"concurrent-meta-reinforcement-learning","title":"Concurrent Meta Reinforcement Learning","date":"2019-03-07","arxiv_id":"1903.02710","repositories_listed":1,"syntology":null},{"url":"/paper/deeper-sparser-exploration","slug":"deeper-sparser-exploration","title":"Bayesian Reinforcement Learning via Deep, Sparse Sampling","date":"2019-02-07","arxiv_id":"1902.02661","repositories_listed":1,"syntology":null},{"url":"/paper/information-directed-exploration-for-deep","slug":"information-directed-exploration-for-deep","title":"Information-Directed Exploration for Deep Reinforcement Learning","date":"2018-12-18","arxiv_id":"1812.07544","repositories_listed":1,"syntology":null},{"url":"/paper/playing-text-adventure-games-with-graph-based","slug":"playing-text-adventure-games-with-graph-based","title":"Playing Text-Adventure Games with Graph-Based Deep Reinforcement Learning","date":"2018-12-04","arxiv_id":"1812.01628","repositories_listed":1,"syntology":null},{"url":"/paper/curious-intrinsically-motivated-multi-task","slug":"curious-intrinsically-motivated-multi-task","title":"CURIOUS: Intrinsically Motivated Modular Multi-Goal Reinforcement Learning","date":"2018-10-15","arxiv_id":"1810.06284","repositories_listed":1,"syntology":null},{"url":"/paper/cm3-cooperative-multi-goal-multi-stage-multi","slug":"cm3-cooperative-multi-goal-multi-stage-multi","title":"CM3: Cooperative Multi-goal Multi-stage Multi-agent Reinforcement Learning","date":"2018-09-13","arxiv_id":"1809.05188","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/cm3-cooperative-multi-goal-multi-stage-multi#ran","syntology_url":"https://syntology.ai/paper/1809.05188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.05188"}},"official":{"repos":["011235813/cm3"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/near-optimal-exploration-exploitation-in-non","slug":"near-optimal-exploration-exploitation-in-non","title":"Near Optimal Exploration-Exploitation in Non-Communicating Markov Decision Processes","date":"2018-07-06","arxiv_id":"1807.02373","repositories_listed":1,"syntology":null},{"url":"/paper/curiosity-driven-exploration-of-learned","slug":"curiosity-driven-exploration-of-learned","title":"Curiosity Driven Exploration of Learned Disentangled Goal Spaces","date":"2018-07-04","arxiv_id":"1807.01521","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-gradient-free-variational-inference","slug":"efficient-gradient-free-variational-inference","title":"Efficient Gradient-Free Variational Inference using Policy Search","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-objective-model-based-policy-search-for","slug":"multi-objective-model-based-policy-search-for","title":"Multi-objective Model-based Policy Search for Data-efficient Learning with Sparse Rewards","date":"2018-06-25","arxiv_id":"1806.09351","repositories_listed":1,"syntology":null},{"url":"/paper/meta-learning-for-stochastic-gradient-mcmc","slug":"meta-learning-for-stochastic-gradient-mcmc","title":"Meta-Learning for Stochastic Gradient MCMC","date":"2018-06-12","arxiv_id":"1806.04522","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-exploration-through-bayesian-deep-q","slug":"efficient-exploration-through-bayesian-deep-q","title":"Efficient Exploration through Bayesian Deep Q-Networks","date":"2018-02-13","arxiv_id":"1802.04412","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-bias-span-constrained-exploration","slug":"efficient-bias-span-constrained-exploration","title":"Efficient Bias-Span-Constrained Exploration-Exploitation in Reinforcement Learning","date":"2018-02-12","arxiv_id":"1802.04020","repositories_listed":1,"syntology":null},{"url":"/paper/federated-control-with-hierarchical-multi","slug":"federated-control-with-hierarchical-multi","title":"Federated Control with Hierarchical Multi-Agent Deep Reinforcement Learning","date":"2017-12-22","arxiv_id":"1712.08266","repositories_listed":1,"syntology":null},{"url":"/paper/variational-deep-q-network","slug":"variational-deep-q-network","title":"Variational Deep Q Network","date":"2017-11-30","arxiv_id":"1711.11225","repositories_listed":1,"syntology":null},{"url":"/paper/count-based-exploration-in-feature-space-for","slug":"count-based-exploration-in-feature-space-for","title":"Count-Based Exploration in Feature Space for Reinforcement Learning","date":"2017-06-25","arxiv_id":"1706.08090","repositories_listed":1,"syntology":null},{"url":"/paper/angrier-birds-bayesian-reinforcement-learning","slug":"angrier-birds-bayesian-reinforcement-learning","title":"Angrier Birds: Bayesian reinforcement learning","date":"2016-01-06","arxiv_id":"1601.01297","repositories_listed":1,"syntology":null},{"url":"/paper/batch-bayesian-optimization-via-local","slug":"batch-bayesian-optimization-via-local","title":"Batch Bayesian Optimization via Local Penalization","date":"2015-05-29","arxiv_id":"1505.08052","repositories_listed":1,"syntology":null},{"url":"/paper/generalization-and-exploration-via-randomized","slug":"generalization-and-exploration-via-randomized","title":"Generalization and Exploration via Randomized Value Functions","date":"2014-02-04","arxiv_id":"1402.0635","repositories_listed":1,"syntology":null},{"url":null,"slug":"moorl-a-framework-for-integrating-offline","title":"MOORL: A Framework for Integrating Offline-Online Reinforcement Learning","date":"2025-06-11","arxiv_id":"2506.09574","repositories_listed":0,"syntology":null},{"url":null,"slug":"drsr-llm-based-scientific-equation-discovery","title":"DrSR: LLM based Scientific Equation Discovery with Dual Reasoning from Data and Experience","date":"2025-06-04","arxiv_id":"2506.04282","repositories_listed":0,"syntology":null},{"url":null,"slug":"go-browse-training-web-agents-with-structured","title":"Go-Browse: Training Web Agents with Structured Exploration","date":"2025-06-04","arxiv_id":"2506.03533","repositories_listed":0,"syntology":null},{"url":null,"slug":"womap-world-models-for-embodied-open","title":"WoMAP: World Models For Embodied Open-Vocabulary Object Localization","date":"2025-06-02","arxiv_id":"2506.01600","repositories_listed":0,"syntology":null},{"url":null,"slug":"helixdesign-binder-a-scalable-production","title":"HelixDesign-Binder: A Scalable Production-Grade Platform for Binder Design Built on HelixFold3","date":"2025-05-28","arxiv_id":"2505.21873","repositories_listed":0,"syntology":null},{"url":null,"slug":"star-r1-spacial-transformation-reasoning-by","title":"STAR-R1: Spacial TrAnsformation Reasoning by Reinforcing Multimodal LLMs","date":"2025-05-21","arxiv_id":"2505.15804","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10843","title":"Comparative Analysis of Black-Box Optimization Methods for Weather Intervention Design","date":"2025-05-16","arxiv_id":"2505.10843","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-ril-interleaved-reinforcement-and","title":"IN-RIL: Interleaved Reinforcement and Imitation Learning for Policy Fine-Tuning","date":"2025-05-15","arxiv_id":"2505.10442","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilling-realizable-students-from","title":"Distilling Realizable Students from Unrealizable Teachers","date":"2025-05-14","arxiv_id":"2505.09546","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-agents-mirror-human-causal-reasoning","title":"Language Agents Mirror Human Causal Reasoning Biases. How Can We Help Them Think Like Scientists?","date":"2025-05-14","arxiv_id":"2505.09614","repositories_listed":0,"syntology":null},{"url":null,"slug":"credit-assignment-and-efficient-exploration","title":"Credit Assignment and Efficient Exploration based on Influence Scope in Multi-agent Reinforcement Learning","date":"2025-05-13","arxiv_id":"2505.08630","repositories_listed":0,"syntology":null}],"record_sha256":"b48f8a99434edc646e90ca30642799e1e05e2f8db898ed2980fa3782aaf02a7d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}