{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/20","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":20,"pages_in_order":152,"rows_per_page":100,"rows":[1901,2000],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/19","next":"/task/reinforcement-learning-1/papers/21","papers":[{"url":"/paper/discovering-hierarchical-achievements-in-1","slug":"discovering-hierarchical-achievements-in-1","title":"Discovering Hierarchical Achievements in Reinforcement Learning via Contrastive Learning","date":"2023-07-07","arxiv_id":"2307.03486","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-observation-policies-in-observation","slug":"dynamic-observation-policies-in-observation","title":"Dynamic Observation Policies in Observation Cost-Sensitive Reinforcement Learning","date":"2023-07-05","arxiv_id":"2307.02620","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/dynamic-observation-policies-in-observation#ran","syntology_url":"https://syntology.ai/paper/2307.02620","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.02620"}},"official":{"repos":["cbellinger27/learning-when-to-observe-in-rl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/first-explore-then-exploit-meta-learning","slug":"first-explore-then-exploit-meta-learning","title":"First-Explore, then Exploit: Meta-Learning to Solve Hard Exploration-Exploitation Trade-Offs","date":"2023-07-05","arxiv_id":"2307.02276","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/first-explore-then-exploit-meta-learning#ran","syntology_url":"https://syntology.ai/paper/2307.02276","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.02276"}},"official":{"repos":["btnorman/First-Explore"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/environmental-effects-on-emergent-strategy-in","slug":"environmental-effects-on-emergent-strategy-in","title":"Environmental effects on emergent strategy in micro-scale multi-agent reinforcement learning","date":"2023-07-03","arxiv_id":"2307.00994","repositories_listed":1,"syntology":null},{"url":"/paper/romo-her-robust-model-based-hindsight","slug":"romo-her-robust-model-based-hindsight","title":"MRHER: Model-based Relay Hindsight Experience Replay for Sequential Object Manipulation Tasks with Sparse Rewards","date":"2023-06-28","arxiv_id":"2306.16061","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-truss-design-with-reinforcement","slug":"automatic-truss-design-with-reinforcement","title":"Automatic Truss Design with Reinforcement Learning","date":"2023-06-27","arxiv_id":"2306.15182","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-modulate-pre-trained-models-in-rl-1","slug":"learning-to-modulate-pre-trained-models-in-rl-1","title":"Learning to Modulate pre-trained Models in RL","date":"2023-06-26","arxiv_id":"2306.14884","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-modulate-pre-trained-models-in-rl-1#ran","syntology_url":"https://syntology.ai/paper/2306.14884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.14884"}},"official":{"repos":["ml-jku/l2m"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multivariate-time-series-early-classification","slug":"multivariate-time-series-early-classification","title":"Multivariate Time Series Early Classification Across Channel and Time Dimensions","date":"2023-06-26","arxiv_id":"2306.14606","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-ood-state-actions-supported-cross","slug":"beyond-ood-state-actions-supported-cross","title":"Beyond OOD State Actions: Supported Cross-Domain Offline Reinforcement Learning","date":"2023-06-22","arxiv_id":"2306.12755","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-ood-state-actions-supported-cross#ran","syntology_url":"https://syntology.ai/paper/2306.12755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.12755"}},"official":{"repos":["thuml/SPOT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/harnessing-mixed-offline-reinforcement","slug":"harnessing-mixed-offline-reinforcement","title":"Harnessing Mixed Offline Reinforcement Learning Datasets via Trajectory Weighting","date":"2023-06-22","arxiv_id":"2306.13085","repositories_listed":1,"syntology":null},{"url":"/paper/taco-temporal-latent-action-driven","slug":"taco-temporal-latent-action-driven","title":"TACO: Temporal Latent Action-Driven Contrastive Loss for Visual Reinforcement Learning","date":"2023-06-22","arxiv_id":"2306.13229","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/taco-temporal-latent-action-driven#ran","syntology_url":"https://syntology.ai/paper/2306.13229","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.13229"}},"official":{"repos":["frankzheng2022/taco"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adcraft-an-advanced-reinforcement-learning","slug":"adcraft-an-advanced-reinforcement-learning","title":"AdCraft: An Advanced Reinforcement Learning Benchmark Environment for Search Engine Marketing Optimization","date":"2023-06-21","arxiv_id":"2306.11971","repositories_listed":1,"syntology":null},{"url":"/paper/state-wise-constrained-policy-optimization","slug":"state-wise-constrained-policy-optimization","title":"State-wise Constrained Policy Optimization","date":"2023-06-21","arxiv_id":"2306.12594","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-search-and-track-with-multiagent","slug":"adversarial-search-and-track-with-multiagent","title":"Adversarial Search and Tracking with Multiagent Reinforcement Learning in Sparsely Observable Environment","date":"2023-06-20","arxiv_id":"2306.11301","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-generate-better-than-your-llm","slug":"learning-to-generate-better-than-your-llm","title":"Learning to Generate Better Than Your LLM","date":"2023-06-20","arxiv_id":"2306.11816","repositories_listed":1,"syntology":null},{"url":"/paper/neural-inventory-control-in-networks-via","slug":"neural-inventory-control-in-networks-via","title":"Neural Inventory Control in Networks via Hindsight Differentiable Policy Optimization","date":"2023-06-20","arxiv_id":"2306.11246","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-ordered-information-extraction-with","slug":"adaptive-ordered-information-extraction-with","title":"Adaptive Ordered Information Extraction with Deep Reinforcement Learning","date":"2023-06-19","arxiv_id":"2306.10787","repositories_listed":1,"syntology":null},{"url":"/paper/adastop-sequential-testing-for-efficient-and","slug":"adastop-sequential-testing-for-efficient-and","title":"AdaStop: adaptive statistical testing for sound comparisons of Deep RL agents","date":"2023-06-19","arxiv_id":"2306.10882","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-generalization-and-plasticity-for","slug":"enhancing-generalization-and-plasticity-for","title":"PLASTIC: Improving Input and Label Plasticity for Sample Efficient Reinforcement Learning","date":"2023-06-19","arxiv_id":"2306.10711","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-quantum-variational-state","slug":"enhancing-quantum-variational-state","title":"Enhancing variational quantum state diagonalization using reinforcement learning techniques","date":"2023-06-19","arxiv_id":"2306.11086","repositories_listed":1,"syntology":null},{"url":"/paper/active-policy-improvement-from-multiple-black","slug":"active-policy-improvement-from-multiple-black","title":"Active Policy Improvement from Multiple Black-box Oracles","date":"2023-06-17","arxiv_id":"2306.10259","repositories_listed":1,"syntology":null},{"url":"/paper/genes-in-intelligent-agents","slug":"genes-in-intelligent-agents","title":"Genes in Intelligent Agents","date":"2023-06-17","arxiv_id":"2306.10225","repositories_listed":1,"syntology":null},{"url":"/paper/jumanji-a-diverse-suite-of-scalable","slug":"jumanji-a-diverse-suite-of-scalable","title":"Jumanji: a Diverse Suite of Scalable Reinforcement Learning Environments in JAX","date":"2023-06-16","arxiv_id":"2306.09884","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/jumanji-a-diverse-suite-of-scalable#ran","syntology_url":"https://syntology.ai/paper/2306.09884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09884"}},"official":{"repos":["instadeepai/jumanji"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/semi-offline-reinforcement-learning-for","slug":"semi-offline-reinforcement-learning-for","title":"Semi-Offline Reinforcement Learning for Optimized Text Generation","date":"2023-06-16","arxiv_id":"2306.09712","repositories_listed":1,"syntology":null},{"url":"/paper/quadswarm-a-modular-multi-quadrotor-simulator","slug":"quadswarm-a-modular-multi-quadrotor-simulator","title":"QuadSwarm: A Modular Multi-Quadrotor Simulator for Deep Reinforcement Learning with Direct Thrust Control","date":"2023-06-15","arxiv_id":"2306.09537","repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-evaluation-in-doubly-inhomogeneous","slug":"off-policy-evaluation-in-doubly-inhomogeneous","title":"Off-policy Evaluation in Doubly Inhomogeneous Environments","date":"2023-06-14","arxiv_id":"2306.08719","repositories_listed":1,"syntology":null},{"url":"/paper/can-chatgpt-enable-its-the-case-of-mixed","slug":"can-chatgpt-enable-its-the-case-of-mixed","title":"Can ChatGPT Enable ITS? The Case of Mixed Traffic Control via Reinforcement Learning","date":"2023-06-13","arxiv_id":"2306.08094","repositories_listed":1,"syntology":null},{"url":"/paper/denselight-efficient-control-for-large-scale","slug":"denselight-efficient-control-for-large-scale","title":"DenseLight: Efficient Control for Large-scale Traffic Signals with Dense Feedback","date":"2023-06-13","arxiv_id":"2306.07553","repositories_listed":1,"syntology":null},{"url":"/paper/galactic-scaling-end-to-end-reinforcement-1","slug":"galactic-scaling-end-to-end-reinforcement-1","title":"Galactic: Scaling End-to-End Reinforcement Learning for Rearrangement at 100k Steps-Per-Second","date":"2023-06-13","arxiv_id":"2306.07552","repositories_listed":1,"syntology":null},{"url":"/paper/online-prototype-alignment-for-few-shot","slug":"online-prototype-alignment-for-few-shot","title":"Online Prototype Alignment for Few-shot Policy Transfer","date":"2023-06-12","arxiv_id":"2306.07307","repositories_listed":1,"syntology":null},{"url":"/paper/transcendental-idealism-of-planner-evaluating","slug":"transcendental-idealism-of-planner-evaluating","title":"Transcendental Idealism of Planner: Evaluating Perception from Planning Perspective for Autonomous Driving","date":"2023-06-12","arxiv_id":"2306.07276","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/transcendental-idealism-of-planner-evaluating#ran","syntology_url":"https://syntology.ai/paper/2306.07276","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.07276"}},"official":{"repos":["qcraftai/tip"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/generalizable-wireless-navigation-through","slug":"generalizable-wireless-navigation-through","title":"Digital Twin-Enhanced Wireless Indoor Navigation: Achieving Efficient Environment Sensing with Zero-Shot Reinforcement Learning","date":"2023-06-11","arxiv_id":"2306.06766","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalizable-wireless-navigation-through#ran","syntology_url":"https://syntology.ai/paper/2306.06766","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.06766"}},"official":{"repos":["panshark/pirl-win"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-end-to-end-reinforcement-learning-approach","slug":"an-end-to-end-reinforcement-learning-approach","title":"An End-to-End Reinforcement Learning Approach for Job-Shop Scheduling Problems Based on Constraint Programming","date":"2023-06-09","arxiv_id":"2306.05747","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/an-end-to-end-reinforcement-learning-approach#ran","syntology_url":"https://syntology.ai/paper/2306.05747","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.05747"}},"official":{"repos":["ingambe/End2End-Job-Shop-Scheduling-CP"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-importance-of-feature-decorrelation","slug":"on-the-importance-of-feature-decorrelation","title":"On the Importance of Feature Decorrelation for Unsupervised Representation Learning in Reinforcement Learning","date":"2023-06-09","arxiv_id":"2306.05637","repositories_listed":1,"syntology":null},{"url":"/paper/look-beneath-the-surface-exploiting-1","slug":"look-beneath-the-surface-exploiting-1","title":"Look Beneath the Surface: Exploiting Fundamental Symmetry for Sample-Efficient Offline RL","date":"2023-06-07","arxiv_id":"2306.04220","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/look-beneath-the-surface-exploiting-1#ran","syntology_url":"https://syntology.ai/paper/2306.04220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.04220"}},"official":{"repos":["pcheng2/tsrl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/backproptools-a-fast-portable-deep","slug":"backproptools-a-fast-portable-deep","title":"RLtools: A Fast, Portable Deep Reinforcement Learning Library for Continuous Control","date":"2023-06-06","arxiv_id":"2306.03530","repositories_listed":1,"syntology":null},{"url":"/paper/mildly-constrained-evaluation-policy-for","slug":"mildly-constrained-evaluation-policy-for","title":"Mildly Constrained Evaluation Policy for Offline Reinforcement Learning","date":"2023-06-06","arxiv_id":"2306.03680","repositories_listed":1,"syntology":null},{"url":"/paper/stabilizing-contrastive-rl-techniques-for","slug":"stabilizing-contrastive-rl-techniques-for","title":"Stabilizing Contrastive RL: Techniques for Robotic Goal Reaching from Offline Data","date":"2023-06-06","arxiv_id":"2306.03346","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/stabilizing-contrastive-rl-techniques-for#ran","syntology_url":"https://syntology.ai/paper/2306.03346","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03346"}},"official":{"repos":["chongyi-zheng/stable_contrastive_rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/vid2act-activate-offline-videos-for-visual-rl","slug":"vid2act-activate-offline-videos-for-visual-rl","title":"Model-Based Reinforcement Learning with Multi-Task Offline Pretraining","date":"2023-06-06","arxiv_id":"2306.03360","repositories_listed":1,"syntology":null},{"url":"/paper/your-value-function-is-a-control-barrier","slug":"your-value-function-is-a-control-barrier","title":"Value Functions are Control Barrier Functions: Verification of Safe Policies using Control Theory","date":"2023-06-06","arxiv_id":"2306.04026","repositories_listed":1,"syntology":null},{"url":"/paper/risk-aware-reward-shaping-of-reinforcement","slug":"risk-aware-reward-shaping-of-reinforcement","title":"Risk-Aware Reward Shaping of Reinforcement Learning Agents for Autonomous Driving","date":"2023-06-05","arxiv_id":"2306.03220","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/risk-aware-reward-shaping-of-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2306.03220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03220"}},"official":{"repos":["zhang-zengjie/code_2023_iecon_shaping_wu"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/seizing-serendipity-exploiting-the-value-of","slug":"seizing-serendipity-exploiting-the-value-of","title":"Seizing Serendipity: Exploiting the Value of Past Success in Off-Policy Actor-Critic","date":"2023-06-05","arxiv_id":"2306.02865","repositories_listed":1,"syntology":null},{"url":"/paper/hyperparameters-in-reinforcement-learning-and","slug":"hyperparameters-in-reinforcement-learning-and","title":"Hyperparameters in Reinforcement Learning and How To Tune Them","date":"2023-06-02","arxiv_id":"2306.01324","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hyperparameters-in-reinforcement-learning-and#ran","syntology_url":"https://syntology.ai/paper/2306.01324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01324"}},"official":{"repos":["facebookresearch/how-to-autorl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tackling-unbounded-state-spaces-in-continuing","slug":"tackling-unbounded-state-spaces-in-continuing","title":"Learning to Stabilize Online Reinforcement Learning in Unbounded State Spaces","date":"2023-06-02","arxiv_id":"2306.01896","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":1,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":2,"n_no_contract":1,"n_pointer_only":6,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 2 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tackling-unbounded-state-spaces-in-continuing#ran","syntology_url":"https://syntology.ai/paper/2306.01896","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01896"}},"official":{"repos":["badger-rl/stop"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/identifiability-and-generalizability-in","slug":"identifiability-and-generalizability-in","title":"Identifiability and Generalizability in Constrained Inverse Reinforcement Learning","date":"2023-06-01","arxiv_id":"2306.00629","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/identifiability-and-generalizability-in#ran","syntology_url":"https://syntology.ai/paper/2306.00629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00629"}},"official":{"repos":["andrschl/cirl"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-and-benchmarking-offline","slug":"improving-and-benchmarking-offline","title":"Improving and Benchmarking Offline Reinforcement Learning Algorithms","date":"2023-06-01","arxiv_id":"2306.00972","repositories_listed":1,"syntology":null},{"url":"/paper/normalization-enhances-generalization-in","slug":"normalization-enhances-generalization-in","title":"Normalization Enhances Generalization in Visual Reinforcement Learning","date":"2023-06-01","arxiv_id":"2306.00656","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/normalization-enhances-generalization-in#ran","syntology_url":"https://syntology.ai/paper/2306.00656","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00656"}},"official":{"repos":["lilucse/Normalization-Enhances-Generalization-in-Visual-Reinforcement-Learning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-offline-reinforcement-learning-with-real","slug":"safe-offline-reinforcement-learning-with-real","title":"Safe Offline Reinforcement Learning with Real-Time Budget Constraints","date":"2023-06-01","arxiv_id":"2306.00603","repositories_listed":1,"syntology":null},{"url":"/paper/thought-cloning-learning-to-think-while-1","slug":"thought-cloning-learning-to-think-while-1","title":"Thought Cloning: Learning to Think while Acting by Imitating Human Thinking","date":"2023-06-01","arxiv_id":"2306.00323","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/thought-cloning-learning-to-think-while-1#ran","syntology_url":"https://syntology.ai/paper/2306.00323","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00323"}},"official":{"repos":["ShengranHu/Thought-Cloning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-diffusion-policies-for-offline-1","slug":"efficient-diffusion-policies-for-offline-1","title":"Efficient Diffusion Policies for Offline Reinforcement Learning","date":"2023-05-31","arxiv_id":"2305.20081","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-diffusion-policies-for-offline-1#ran","syntology_url":"https://syntology.ai/paper/2305.20081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.20081"}},"official":{"repos":["sail-sg/edp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-for-edge-weighted-online-bipartite","slug":"learning-for-edge-weighted-online-bipartite","title":"Learning for Edge-Weighted Online Bipartite Matching with Robustness Guarantees","date":"2023-05-31","arxiv_id":"2306.00172","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/learning-for-edge-weighted-online-bipartite#ran","syntology_url":"https://syntology.ai/paper/2306.00172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00172"}},"official":{"repos":["ren-research/lomar"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-reinforcement-learning-objectives-for","slug":"robust-reinforcement-learning-objectives-for","title":"Robust Reinforcement Learning Objectives for Sequential Recommender Systems","date":"2023-05-30","arxiv_id":"2305.18820","repositories_listed":1,"syntology":null},{"url":"/paper/subequivariant-graph-reinforcement-learning","slug":"subequivariant-graph-reinforcement-learning","title":"Subequivariant Graph Reinforcement Learning in 3D Environments","date":"2023-05-30","arxiv_id":"2305.18951","repositories_listed":1,"syntology":{"n":18,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":13,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/subequivariant-graph-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2305.18951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18951"}},"official":{"repos":["alpc91/sgrl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":13,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-model-is-an-effective-planner-and-1","slug":"diffusion-model-is-an-effective-planner-and-1","title":"Diffusion Model is an Effective Planner and Data Synthesizer for Multi-Task Reinforcement Learning","date":"2023-05-29","arxiv_id":"2305.18459","repositories_listed":1,"syntology":{"n":30,"n_ran":19,"n_constructed":3,"n_ran_checked":18,"n_instrument":1,"n_unverified":11,"n_honours":5,"n_violates":0,"n_no_contract":13,"n_pointer_only":4,"phrase":"19 ran (of which 3 constructed an object rather than computing a result; 18 with no instrument failure: 5 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/diffusion-model-is-an-effective-planner-and-1#ran","syntology_url":"https://syntology.ai/paper/2305.18459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18459"}},"official":{"repos":["tinnerhrhe/MTDiff"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":7,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/off-policy-rl-algorithms-can-be-sample","slug":"off-policy-rl-algorithms-can-be-sample","title":"Off-Policy RL Algorithms Can be Sample-Efficient for Continuous Control via Sample Multiple Reuse","date":"2023-05-29","arxiv_id":"2305.18443","repositories_listed":1,"syntology":null},{"url":"/paper/privileged-knowledge-distillation-for-sim-to","slug":"privileged-knowledge-distillation-for-sim-to","title":"Bridging the Sim-to-Real Gap from the Information Bottleneck Perspective","date":"2023-05-29","arxiv_id":"2305.18464","repositories_listed":1,"syntology":null},{"url":"/paper/provable-and-practical-efficient-exploration","slug":"provable-and-practical-efficient-exploration","title":"Provable and Practical: Efficient Exploration in Reinforcement Learning via Langevin Monte Carlo","date":"2023-05-29","arxiv_id":"2305.18246","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/provable-and-practical-efficient-exploration#ran","syntology_url":"https://syntology.ai/paper/2305.18246","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18246"}},"official":{"repos":["hmishfaq/lmc-lsvi"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/madiff-offline-multi-agent-learning-with","slug":"madiff-offline-multi-agent-learning-with","title":"MADiff: Offline Multi-agent Learning with Diffusion Models","date":"2023-05-27","arxiv_id":"2305.17330","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/madiff-offline-multi-agent-learning-with#ran","syntology_url":"https://syntology.ai/paper/2305.17330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17330"}},"official":{"repos":["zbzhu99/madiff"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-reminder-of-its-brittleness-language-reward","slug":"a-reminder-of-its-brittleness-language-reward","title":"A Reminder of its Brittleness: Language Reward Shaping May Hinder Learning for Instruction Following Agents","date":"2023-05-26","arxiv_id":"2305.16621","repositories_listed":1,"syntology":null},{"url":"/paper/future-conditioned-unsupervised-pretraining","slug":"future-conditioned-unsupervised-pretraining","title":"Future-conditioned Unsupervised Pretraining for Decision Transformer","date":"2023-05-26","arxiv_id":"2305.16683","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/future-conditioned-unsupervised-pretraining#ran","syntology_url":"https://syntology.ai/paper/2305.16683","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.16683"}},"official":{"repos":["fffffarmer/pdt"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/generating-synergistic-formulaic-alpha","slug":"generating-synergistic-formulaic-alpha","title":"Generating Synergistic Formulaic Alpha Collections via Reinforcement Learning","date":"2023-05-25","arxiv_id":"2306.12964","repositories_listed":1,"syntology":null},{"url":"/paper/ghost-in-the-minecraft-generally-capable","slug":"ghost-in-the-minecraft-generally-capable","title":"Ghost in the Minecraft: Generally Capable Agents for Open-World Environments via Large Language Models with Text-based Knowledge and Memory","date":"2023-05-25","arxiv_id":"2305.17144","repositories_listed":1,"syntology":null},{"url":"/paper/market-making-with-deep-reinforcement","slug":"market-making-with-deep-reinforcement","title":"Market Making with Deep Reinforcement Learning from Limit Order Books","date":"2023-05-25","arxiv_id":"2305.15821","repositories_listed":1,"syntology":null},{"url":"/paper/proto-iterative-policy-regularized-offline-to","slug":"proto-iterative-policy-regularized-offline-to","title":"PROTO: Iterative Policy Regularized Offline-to-Online Reinforcement Learning","date":"2023-05-25","arxiv_id":"2305.15669","repositories_listed":1,"syntology":null},{"url":"/paper/reward-machine-guided-self-paced","slug":"reward-machine-guided-self-paced","title":"Reward-Machine-Guided, Self-Paced Reinforcement Learning","date":"2023-05-25","arxiv_id":"2305.16505","repositories_listed":1,"syntology":null},{"url":"/paper/collaborative-world-models-an-online-offline","slug":"collaborative-world-models-an-online-offline","title":"Making Offline RL Online: Collaborative World Models for Offline Visual Reinforcement Learning","date":"2023-05-24","arxiv_id":"2305.15260","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":7,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":13,"phrase":"10 ran (of which 7 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/collaborative-world-models-an-online-offline#ran","syntology_url":"https://syntology.ai/paper/2305.15260","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15260"}},"official":{"repos":["qiwang067/CoWorld"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":7,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/decision-aware-actor-critic-with-function-1","slug":"decision-aware-actor-critic-with-function-1","title":"Decision-Aware Actor-Critic with Function Approximation and Theoretical Guarantees","date":"2023-05-24","arxiv_id":"2305.15249","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/decision-aware-actor-critic-with-function-1#ran","syntology_url":"https://syntology.ai/paper/2305.15249","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15249"}},"official":{"repos":["amirrezakazemi/acpg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/spring-gpt-4-out-performs-rl-algorithms-by","slug":"spring-gpt-4-out-performs-rl-algorithms-by","title":"SPRING: Studying the Paper and Reasoning to Play Games","date":"2023-05-24","arxiv_id":"2305.15486","repositories_listed":1,"syntology":null},{"url":"/paper/2305-14550","slug":"2305-14550","title":"When should we prefer Decision Transformers for Offline Reinforcement Learning?","date":"2023-05-23","arxiv_id":"2305.14550","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/2305-14550#ran","syntology_url":"https://syntology.ai/paper/2305.14550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14550"}},"official":{"repos":["prajjwal1/rl_paradigm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/conditional-mutual-information-for-1","slug":"conditional-mutual-information-for-1","title":"Conditional Mutual Information for Disentangled Representations in Reinforcement Learning","date":"2023-05-23","arxiv_id":"2305.14133","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conditional-mutual-information-for-1#ran","syntology_url":"https://syntology.ai/paper/2305.14133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14133"}},"official":{"repos":["uoe-agents/cmid"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/furniturebench-reproducible-real-world","slug":"furniturebench-reproducible-real-world","title":"FurnitureBench: Reproducible Real-World Benchmark for Long-Horizon Complex Manipulation","date":"2023-05-22","arxiv_id":"2305.12821","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/furniturebench-reproducible-real-world#ran","syntology_url":"https://syntology.ai/paper/2305.12821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12821"}},"official":{"repos":["clvrai/furniture-bench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-representation-via-diffusion","slug":"policy-representation-via-diffusion","title":"Policy Representation via Diffusion Probability Model for Reinforcement Learning","date":"2023-05-22","arxiv_id":"2305.13122","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/policy-representation-via-diffusion#ran","syntology_url":"https://syntology.ai/paper/2305.13122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13122"}},"official":{"repos":["bellmantimehut/dipo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/regularization-and-variance-weighted","slug":"regularization-and-variance-weighted","title":"Regularization and Variance-Weighted Regression Achieves Minimax Optimality in Linear MDPs: Theory and Practice","date":"2023-05-22","arxiv_id":"2305.13185","repositories_listed":1,"syntology":null},{"url":"/paper/bertrlfuzzer-a-bert-and-reinforcement","slug":"bertrlfuzzer-a-bert-and-reinforcement","title":"BertRLFuzzer: A BERT and Reinforcement Learning Based Fuzzer","date":"2023-05-21","arxiv_id":"2305.12534","repositories_listed":1,"syntology":null},{"url":"/paper/sneakyprompt-evaluating-robustness-of-text-to","slug":"sneakyprompt-evaluating-robustness-of-text-to","title":"SneakyPrompt: Jailbreaking Text-to-image Generative Models","date":"2023-05-20","arxiv_id":"2305.12082","repositories_listed":1,"syntology":null},{"url":"/paper/learning-diverse-risk-preferences-in","slug":"learning-diverse-risk-preferences-in","title":"Learning Diverse Risk Preferences in Population-based Self-play","date":"2023-05-19","arxiv_id":"2305.11476","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-diverse-risk-preferences-in#ran","syntology_url":"https://syntology.ai/paper/2305.11476","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11476"}},"official":{"repos":["jackory/rpbt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/client-selection-for-federated-policy","slug":"client-selection-for-federated-policy","title":"Client Selection for Federated Policy Optimization with Environment Heterogeneity","date":"2023-05-18","arxiv_id":"2305.10978","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-state-augmentations-for","slug":"contrastive-state-augmentations-for","title":"Contrastive State Augmentations for Reinforcement Learning-Based Recommender Systems","date":"2023-05-18","arxiv_id":"2305.11081","repositories_listed":1,"syntology":null},{"url":"/paper/a-genetic-fuzzy-system-for-interpretable-and","slug":"a-genetic-fuzzy-system-for-interpretable-and","title":"A Genetic Fuzzy System for Interpretable and Parsimonious Reinforcement Learning Policies","date":"2023-05-17","arxiv_id":"2305.09922","repositories_listed":1,"syntology":null},{"url":"/paper/demonstration-free-autonomous-reinforcement","slug":"demonstration-free-autonomous-reinforcement","title":"Demonstration-free Autonomous Reinforcement Learning via Implicit and Bidirectional Curriculum","date":"2023-05-17","arxiv_id":"2305.09943","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":1,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":8,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/demonstration-free-autonomous-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2305.09943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.09943"}},"official":{"repos":["snu-larr/ibc_official"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/pittsburgh-learning-classifier-systems-for","slug":"pittsburgh-learning-classifier-systems-for","title":"Pittsburgh Learning Classifier Systems for Explainable Reinforcement Learning: Comparing with XCS","date":"2023-05-17","arxiv_id":"2305.09945","repositories_listed":1,"syntology":null},{"url":"/paper/omnisafe-an-infrastructure-for-accelerating","slug":"omnisafe-an-infrastructure-for-accelerating","title":"OmniSafe: An Infrastructure for Accelerating Safe Reinforcement Learning Research","date":"2023-05-16","arxiv_id":"2305.09304","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based-exploration","slug":"deep-reinforcement-learning-based-exploration","title":"Deep Reinforcement Learning-based Exploration of Web Applications","date":"2023-05-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/what-matters-in-reinforcement-learning-for","slug":"what-matters-in-reinforcement-learning-for","title":"What Matters in Reinforcement Learning for Tractography","date":"2023-05-15","arxiv_id":"2305.09041","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-resources","slug":"multi-agent-reinforcement-learning-resources","title":"Multi-Agent Reinforcement Learning Resources Allocation Method Using Dueling Double Deep Q-Network in Vehicular Networks","date":"2023-05-12","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/quantile-based-deep-reinforcement-learning","slug":"quantile-based-deep-reinforcement-learning","title":"Quantile-Based Deep Reinforcement Learning using Two-Timescale Policy Gradient Algorithms","date":"2023-05-12","arxiv_id":"2305.07248","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-energy-system-scheduling-using-a","slug":"optimal-energy-system-scheduling-using-a","title":"Optimal Energy System Scheduling Using A Constraint-Aware Reinforcement Learning Algorithm","date":"2023-05-09","arxiv_id":"2305.05484","repositories_listed":1,"syntology":null},{"url":"/paper/smaclite-a-lightweight-environment-for-multi","slug":"smaclite-a-lightweight-environment-for-multi","title":"SMAClite: A Lightweight Environment for Multi-Agent Reinforcement Learning","date":"2023-05-09","arxiv_id":"2305.05566","repositories_listed":1,"syntology":null},{"url":"/paper/information-design-in-multi-agent","slug":"information-design-in-multi-agent","title":"Information Design in Multi-Agent Reinforcement Learning","date":"2023-05-08","arxiv_id":"2305.06807","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-topic-models","slug":"reinforcement-learning-for-topic-models","title":"Reinforcement Learning for Topic Models","date":"2023-05-08","arxiv_id":"2305.04843","repositories_listed":1,"syntology":null},{"url":"/paper/explainable-reinforcement-learning-via-a","slug":"explainable-reinforcement-learning-via-a","title":"Explainable Reinforcement Learning via a Causal World Model","date":"2023-05-04","arxiv_id":"2305.02749","repositories_listed":1,"syntology":null},{"url":"/paper/federated-ensemble-directed-offline","slug":"federated-ensemble-directed-offline","title":"Federated Ensemble-Directed Offline Reinforcement Learning","date":"2023-05-04","arxiv_id":"2305.03097","repositories_listed":1,"syntology":null},{"url":"/paper/imap-intrinsically-motivated-adversarial","slug":"imap-intrinsically-motivated-adversarial","title":"Toward Evaluating Robustness of Reinforcement Learning with Adversarial Policy","date":"2023-05-04","arxiv_id":"2305.02605","repositories_listed":1,"syntology":null},{"url":"/paper/simple-noisy-environment-augmentation-for","slug":"simple-noisy-environment-augmentation-for","title":"Simple Noisy Environment Augmentation for Reinforcement Learning","date":"2023-05-04","arxiv_id":"2305.02882","repositories_listed":1,"syntology":null},{"url":"/paper/sim2rec-a-simulator-based-decision-making","slug":"sim2rec-a-simulator-based-decision-making","title":"Sim2Rec: A Simulator-based Decision-making Approach to Optimize Real-World Long-term User Engagement in Sequential Recommender Systems","date":"2023-05-03","arxiv_id":"2305.04832","repositories_listed":1,"syntology":null},{"url":"/paper/an-autonomous-non-monolithic-agent-with-multi","slug":"an-autonomous-non-monolithic-agent-with-multi","title":"An Autonomous Non-monolithic Agent with Multi-mode Exploration based on Options Framework","date":"2023-05-02","arxiv_id":"2305.01322","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-model-free-reinforcement-1","slug":"sample-efficient-model-free-reinforcement-1","title":"Sample Efficient Model-free Reinforcement Learning from LTL Specifications with Optimality Guarantees","date":"2023-05-02","arxiv_id":"2305.01381","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/sample-efficient-model-free-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2305.01381","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.01381"}},"official":{"repos":["shaodaqian/rl-from-ltl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/online-portfolio-management-via-deep","slug":"online-portfolio-management-via-deep","title":"Online Portfolio Management via Deep Reinforcement Learning with High-Frequency Data","date":"2023-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/x-rlflow-graph-reinforcement-learning-for","slug":"x-rlflow-graph-reinforcement-learning-for","title":"X-RLflow: Graph Reinforcement Learning for Neural Network Subgraphs Transformation","date":"2023-04-28","arxiv_id":"2304.14698","repositories_listed":1,"syntology":null},{"url":"/paper/crop-towards-distributional-shift-robust","slug":"crop-towards-distributional-shift-robust","title":"CROP: Towards Distributional-Shift Robust Reinforcement Learning using Compact Reshaped Observation Processing","date":"2023-04-26","arxiv_id":"2304.13616","repositories_listed":1,"syntology":null}],"record_sha256":"b0c8397b5bda10b84603faee4bc9eb8323e055796691b997002cb5a9ff630921","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}