{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/23","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":23,"pages_in_order":135,"rows_per_page":100,"rows":[2201,2300],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/22","next":"/task/reinforcement-learning-2/papers/24","papers":[{"url":"/paper/hierarchical-adversarial-inverse","slug":"hierarchical-adversarial-inverse","title":"Option-Aware Adversarial Inverse Reinforcement Learning for Robotic Control","date":"2022-10-05","arxiv_id":"2210.01969","repositories_listed":1,"syntology":null},{"url":"/paper/towards-safe-mechanical-ventilation-treatment","slug":"towards-safe-mechanical-ventilation-treatment","title":"Towards Safe Mechanical Ventilation Treatment Using Deep Offline Reinforcement Learning","date":"2022-10-05","arxiv_id":"2210.02552","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-safe-mechanical-ventilation-treatment#ran","syntology_url":"https://syntology.ai/paper/2210.02552","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.02552"}},"official":{"repos":["FlemmingKondrup/DeepVent"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/discover-deep-identification-of-symbolic-open","slug":"discover-deep-identification-of-symbolic-open","title":"DISCOVER: Deep identification of symbolically concise open-form PDEs via enhanced reinforcement-learning","date":"2022-10-04","arxiv_id":"2210.02181","repositories_listed":1,"syntology":null},{"url":"/paper/accelerate-reinforcement-learning-with-pid","slug":"accelerate-reinforcement-learning-with-pid","title":"Accelerate Reinforcement Learning with PID Controllers in the Pendulum Simulations","date":"2022-10-03","arxiv_id":"2210.00770","repositories_listed":1,"syntology":null},{"url":"/paper/cairl-a-high-performance-reinforcement","slug":"cairl-a-high-performance-reinforcement","title":"CaiRL: A High-Performance Reinforcement Learning Environment Toolkit","date":"2022-10-03","arxiv_id":"2210.01235","repositories_listed":1,"syntology":null},{"url":"/paper/safe-reinforcement-learning-from-pixels-using","slug":"safe-reinforcement-learning-from-pixels-using","title":"Safe Reinforcement Learning From Pixels Using a Stochastic Latent Representation","date":"2022-10-02","arxiv_id":"2210.01801","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safe-reinforcement-learning-from-pixels-using#ran","syntology_url":"https://syntology.ai/paper/2210.01801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.01801"}},"official":{"repos":["safe-slac/safe-slac"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/b2rl-an-open-source-dataset-for-building","slug":"b2rl-an-open-source-dataset-for-building","title":"B2RL: An open-source Dataset for Building Batch Reinforcement Learning","date":"2022-09-30","arxiv_id":"2209.15626","repositories_listed":1,"syntology":null},{"url":"/paper/hyperbolic-vae-via-latent-gaussian-1","slug":"hyperbolic-vae-via-latent-gaussian-1","title":"Hyperbolic VAE via Latent Gaussian Distributions","date":"2022-09-30","arxiv_id":"2209.15217","repositories_listed":1,"syntology":null},{"url":"/paper/s2p-state-conditioned-image-synthesis-for","slug":"s2p-state-conditioned-image-synthesis-for","title":"S2P: State-conditioned Image Synthesis for Data Augmentation in Offline Reinforcement Learning","date":"2022-09-30","arxiv_id":"2209.15256","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/s2p-state-conditioned-image-synthesis-for#ran","syntology_url":"https://syntology.ai/paper/2209.15256","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.15256"}},"official":{"repos":["dsshim0125/s2p"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-exploration-method-for-reinforcement","slug":"safe-exploration-method-for-reinforcement","title":"Safe Exploration Method for Reinforcement Learning under Existence of Disturbance","date":"2022-09-30","arxiv_id":"2209.15452","repositories_listed":1,"syntology":null},{"url":"/paper/does-zero-shot-reinforcement-learning-exist","slug":"does-zero-shot-reinforcement-learning-exist","title":"Does Zero-Shot Reinforcement Learning Exist?","date":"2022-09-29","arxiv_id":"2209.14935","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/does-zero-shot-reinforcement-learning-exist#ran","syntology_url":"https://syntology.ai/paper/2209.14935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.14935"}},"official":{"repos":["facebookresearch/controllable_agent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-reinforcement-learning-via-high","slug":"offline-reinforcement-learning-via-high","title":"Offline Reinforcement Learning via High-Fidelity Generative Behavior Modeling","date":"2022-09-29","arxiv_id":"2209.14548","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-laws-for-a-multi-agent-reinforcement","slug":"scaling-laws-for-a-multi-agent-reinforcement","title":"Scaling Laws for a Multi-Agent Reinforcement Learning Model","date":"2022-09-29","arxiv_id":"2210.00849","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-laws-for-a-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2210.00849","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.00849"}},"official":{"repos":["orenneumann/alphazero-scaling-laws"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/optimistic-posterior-sampling-for-1","slug":"optimistic-posterior-sampling-for-1","title":"Optimistic Posterior Sampling for Reinforcement Learning with Few Samples and Tight Guarantees","date":"2022-09-28","arxiv_id":"2209.14414","repositories_listed":1,"syntology":null},{"url":"/paper/pareto-actor-critic-for-equilibrium-selection","slug":"pareto-actor-critic-for-equilibrium-selection","title":"Pareto Actor-Critic for Equilibrium Selection in Multi-Agent Reinforcement Learning","date":"2022-09-28","arxiv_id":"2209.14344","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-tensor-networks","slug":"reinforcement-learning-with-tensor-networks","title":"Combining Reinforcement Learning and Tensor Networks, with an Application to Dynamical Large Deviations","date":"2022-09-28","arxiv_id":"2209.14089","repositories_listed":1,"syntology":null},{"url":"/paper/exploiting-transformer-in-reinforcement","slug":"exploiting-transformer-in-reinforcement","title":"Exploiting Transformer in Sparse Reward Reinforcement Learning for Interpretable Temporal Logic Motion Planning","date":"2022-09-27","arxiv_id":"2209.13220","repositories_listed":1,"syntology":null},{"url":"/paper/enhanced-meta-reinforcement-learning-using","slug":"enhanced-meta-reinforcement-learning-using","title":"Enhanced Meta Reinforcement Learning using Demonstrations in Sparse Reward Environments","date":"2022-09-26","arxiv_id":"2209.13048","repositories_listed":1,"syntology":null},{"url":"/paper/explainable-reinforcement-learning-via-model","slug":"explainable-reinforcement-learning-via-model","title":"Explainable Reinforcement Learning via Model Transforms","date":"2022-09-24","arxiv_id":"2209.12006","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"0 ran · 3 unverified","sample_list":"/paper/explainable-reinforcement-learning-via-model#ran","syntology_url":"https://syntology.ai/paper/2209.12006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.12006"}},"official":{"repos":["sarah-keren/rlpe"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/unsupervised-model-based-pre-training-for","slug":"unsupervised-model-based-pre-training-for","title":"Mastering the Unsupervised Reinforcement Learning Benchmark from Pixels","date":"2022-09-24","arxiv_id":"2209.12016","repositories_listed":1,"syntology":{"n":24,"n_ran":17,"n_constructed":14,"n_ran_checked":15,"n_instrument":2,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"17 ran (of which 14 constructed an object rather than computing a result; 15 with no instrument failure: 1 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/unsupervised-model-based-pre-training-for#ran","syntology_url":"https://syntology.ai/paper/2209.12016","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.12016"}},"official":{"repos":["mazpie/mastering-urlb"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":14,"n_ran_no_instrument_failure":15,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/identifiability-and-generalizability-from","slug":"identifiability-and-generalizability-from","title":"Identifiability and generalizability from multiple experts in Inverse Reinforcement Learning","date":"2022-09-22","arxiv_id":"2209.10974","repositories_listed":1,"syntology":null},{"url":"/paper/pretraining-the-vision-transformer-using-self","slug":"pretraining-the-vision-transformer-using-self","title":"Pretraining the Vision Transformer using self-supervised methods for vision based Deep Reinforcement Learning","date":"2022-09-22","arxiv_id":"2209.10901","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-decentralized-deep-reinforcement","slug":"hierarchical-decentralized-deep-reinforcement","title":"Hierarchical Decentralized Deep Reinforcement Learning Architecture for a Simulated Four-Legged Agent","date":"2022-09-21","arxiv_id":"2210.08003","repositories_listed":1,"syntology":null},{"url":"/paper/lcrl-certified-policy-synthesis-via-logically","slug":"lcrl-certified-policy-synthesis-via-logically","title":"LCRL: Certified Policy Synthesis via Logically-Constrained Reinforcement Learning","date":"2022-09-21","arxiv_id":"2209.10341","repositories_listed":1,"syntology":null},{"url":"/paper/a-joint-imitation-reinforcement-learning","slug":"a-joint-imitation-reinforcement-learning","title":"A Joint Imitation-Reinforcement Learning Framework for Reduced Baseline Regret","date":"2022-09-20","arxiv_id":"2209.09446","repositories_listed":1,"syntology":null},{"url":"/paper/latent-plans-for-task-agnostic-offline","slug":"latent-plans-for-task-agnostic-offline","title":"Latent Plans for Task-Agnostic Offline Reinforcement Learning","date":"2022-09-19","arxiv_id":"2209.08959","repositories_listed":1,"syntology":null},{"url":"/paper/man-multi-action-networks-learning","slug":"man-multi-action-networks-learning","title":"MAN: Multi-Action Networks Learning","date":"2022-09-19","arxiv_id":"2209.09329","repositories_listed":1,"syntology":null},{"url":"/paper/measuring-interventional-robustness-in","slug":"measuring-interventional-robustness-in","title":"Measuring Interventional Robustness in Reinforcement Learning","date":"2022-09-19","arxiv_id":"2209.09058","repositories_listed":1,"syntology":null},{"url":"/paper/honor-of-kings-arena-an-environment-for","slug":"honor-of-kings-arena-an-environment-for","title":"Honor of Kings Arena: an Environment for Generalization in Competitive Reinforcement Learning","date":"2022-09-18","arxiv_id":"2209.08483","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/honor-of-kings-arena-an-environment-for#ran","syntology_url":"https://syntology.ai/paper/2209.08483","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.08483"}},"official":{"repos":["tencent-ailab/hok_env"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-natural-language-generation-for-task","slug":"adaptive-natural-language-generation-for-task","title":"Adaptive Natural Language Generation for Task-oriented Dialogue via Reinforcement Learning","date":"2022-09-16","arxiv_id":"2209.07873","repositories_listed":1,"syntology":null},{"url":"/paper/look-where-you-look-saliency-guided-q","slug":"look-where-you-look-saliency-guided-q","title":"Look where you look! Saliency-guided Q-networks for generalization in visual Reinforcement Learning","date":"2022-09-16","arxiv_id":"2209.09203","repositories_listed":1,"syntology":null},{"url":"/paper/m-2-dqn-a-robust-method-for-accelerating-deep","slug":"m-2-dqn-a-robust-method-for-accelerating-deep","title":"M$^2$DQN: A Robust Method for Accelerating Deep Q-learning Network","date":"2022-09-16","arxiv_id":"2209.07809","repositories_listed":1,"syntology":null},{"url":"/paper/stability-constrained-reinforcement-learning-1","slug":"stability-constrained-reinforcement-learning-1","title":"Stability Constrained Reinforcement Learning for Decentralized Real-Time Voltage Control","date":"2022-09-16","arxiv_id":"2209.07669","repositories_listed":1,"syntology":null},{"url":"/paper/toward-safe-and-accelerated-deep","slug":"toward-safe-and-accelerated-deep","title":"Toward Safe and Accelerated Deep Reinforcement Learning for Next-Generation Wireless Networks","date":"2022-09-16","arxiv_id":"2209.13532","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-reuse-bias-in-off-policy-reinforcement","slug":"on-the-reuse-bias-in-off-policy-reinforcement","title":"On the Reuse Bias in Off-Policy Reinforcement Learning","date":"2022-09-15","arxiv_id":"2209.07074","repositories_listed":1,"syntology":null},{"url":"/paper/learning-state-correspondence-of","slug":"learning-state-correspondence-of","title":"Knowledge Transfer in Deep Reinforcement Learning via an RL-Specific GAN-Based Correspondence Function","date":"2022-09-14","arxiv_id":"2209.06604","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-1","slug":"deep-reinforcement-learning-for-1","title":"Deep Reinforcement Learning for Cryptocurrency Trading: Practical Approach to Address Backtest Overfitting","date":"2022-09-12","arxiv_id":"2209.05559","repositories_listed":1,"syntology":{"n":13,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/deep-reinforcement-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/2209.05559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.05559"}},"official":null}},{"url":"/paper/model-based-reinforcement-learning-with-multi","slug":"model-based-reinforcement-learning-with-multi","title":"Model-based Reinforcement Learning with Multi-step Plan Value Estimation","date":"2022-09-12","arxiv_id":"2209.05530","repositories_listed":1,"syntology":null},{"url":"/paper/aerial-view-goal-localization-with","slug":"aerial-view-goal-localization-with","title":"Aerial View Localization with Reinforcement Learning: Towards Emulating Search-and-Rescue","date":"2022-09-08","arxiv_id":"2209.03694","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"0 ran · 3 unverified","sample_list":"/paper/aerial-view-goal-localization-with#ran","syntology_url":"https://syntology.ai/paper/2209.03694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.03694"}},"official":{"repos":["aleksispi/airloc"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/an-empirical-evaluation-of-posterior-sampling","slug":"an-empirical-evaluation-of-posterior-sampling","title":"An Empirical Evaluation of Posterior Sampling for Constrained Reinforcement Learning","date":"2022-09-08","arxiv_id":"2209.03596","repositories_listed":1,"syntology":null},{"url":"/paper/reward-delay-attacks-on-deep-reinforcement","slug":"reward-delay-attacks-on-deep-reinforcement","title":"Reward Delay Attacks on Deep Reinforcement Learning","date":"2022-09-08","arxiv_id":"2209.03540","repositories_listed":1,"syntology":null},{"url":"/paper/hearts-gym-learning-reinforcement-learning-as","slug":"hearts-gym-learning-reinforcement-learning-as","title":"Hearts Gym: Learning Reinforcement Learning as a Team Event","date":"2022-09-07","arxiv_id":"2209.05466","repositories_listed":1,"syntology":null},{"url":"/paper/annealing-optimization-for-progressive","slug":"annealing-optimization-for-progressive","title":"Annealing Optimization for Progressive Learning with Stochastic Approximation","date":"2022-09-06","arxiv_id":"2209.02826","repositories_listed":1,"syntology":null},{"url":"/paper/project-proposal-a-modular-reinforcement","slug":"project-proposal-a-modular-reinforcement","title":"Project proposal: A modular reinforcement learning based automated theorem prover","date":"2022-09-06","arxiv_id":"2209.02562","repositories_listed":1,"syntology":null},{"url":"/paper/intrinsic-fluctuations-of-reinforcement","slug":"intrinsic-fluctuations-of-reinforcement","title":"Intrinsic fluctuations of reinforcement learning promote cooperation","date":"2022-09-01","arxiv_id":"2209.01013","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-conversational-recommendations-is","slug":"rethinking-conversational-recommendations-is","title":"Rethinking Conversational Recommendations: Is Decision Tree All You Need?","date":"2022-08-31","arxiv_id":"2208.14614","repositories_listed":1,"syntology":null},{"url":"/paper/style-agnostic-reinforcement-learning","slug":"style-agnostic-reinforcement-learning","title":"Style-Agnostic Reinforcement Learning","date":"2022-08-31","arxiv_id":"2208.14863","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/style-agnostic-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2208.14863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.14863"}},"official":{"repos":["postech-cvlab/style-agnostic-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/effective-multi-user-delay-constrained","slug":"effective-multi-user-delay-constrained","title":"Effective Multi-User Delay-Constrained Scheduling with Deep Recurrent Reinforcement Learning","date":"2022-08-30","arxiv_id":"2208.14074","repositories_listed":1,"syntology":null},{"url":"/paper/goal-conditioned-q-learning-as-knowledge","slug":"goal-conditioned-q-learning-as-knowledge","title":"Goal-Conditioned Q-Learning as Knowledge Distillation","date":"2022-08-28","arxiv_id":"2208.13298","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-representation-learning-in-deep","slug":"unsupervised-representation-learning-in-deep","title":"Unsupervised Representation Learning in Deep Reinforcement Learning: A Review","date":"2022-08-27","arxiv_id":"2208.14226","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unsupervised-representation-learning-in-deep#ran","syntology_url":"https://syntology.ai/paper/2208.14226","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.14226"}},"official":{"repos":["nicob15/state_representation_learning_methods"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-comparison-of-reinforcement-learning-1","slug":"a-comparison-of-reinforcement-learning-1","title":"A Comparison of Reinforcement Learning Frameworks for Software Testing Tasks","date":"2022-08-25","arxiv_id":"2208.12136","repositories_listed":1,"syntology":null},{"url":"/paper/light-weight-probing-of-unsupervised","slug":"light-weight-probing-of-unsupervised","title":"Light-weight probing of unsupervised representations for Reinforcement Learning","date":"2022-08-25","arxiv_id":"2208.12345","repositories_listed":1,"syntology":null},{"url":"/paper/get-it-in-writing-formal-contracts-mitigate","slug":"get-it-in-writing-formal-contracts-mitigate","title":"Formal Contracts Mitigate Social Dilemmas in Multi-Agent RL","date":"2022-08-22","arxiv_id":"2208.10469","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/get-it-in-writing-formal-contracts-mitigate#ran","syntology_url":"https://syntology.ai/paper/2208.10469","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.10469"}},"official":{"repos":["algorithmic-alignment-lab/contracts"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-post-processing-of-audio-event","slug":"improving-post-processing-of-audio-event","title":"Improving Post-Processing of Audio Event Detectors Using Reinforcement Learning","date":"2022-08-19","arxiv_id":"2208.09201","repositories_listed":1,"syntology":null},{"url":"/paper/a-walk-in-the-park-learning-to-walk-in-20","slug":"a-walk-in-the-park-learning-to-walk-in-20","title":"A Walk in the Park: Learning to Walk in 20 Minutes With Model-Free Reinforcement Learning","date":"2022-08-16","arxiv_id":"2208.07860","repositories_listed":1,"syntology":null},{"url":"/paper/pd-morl-preference-driven-multi-objective","slug":"pd-morl-preference-driven-multi-objective","title":"PD-MORL: Preference-Driven Multi-Objective Reinforcement Learning Algorithm","date":"2022-08-16","arxiv_id":"2208.07914","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/pd-morl-preference-driven-multi-objective#ran","syntology_url":"https://syntology.ai/paper/2208.07914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.07914"}},"official":{"repos":["tbasaklar/PDMORL-Preference-Driven-Multi-Objective-Reinforcement-Learning-Algorithm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-design-for-an-online-reinforcement","slug":"reward-design-for-an-online-reinforcement","title":"Reward Design For An Online Reinforcement Learning Algorithm Supporting Oral Self-Care","date":"2022-08-15","arxiv_id":"2208.07406","repositories_listed":1,"syntology":null},{"url":"/paper/transformer-based-value-function","slug":"transformer-based-value-function","title":"Transformer-based Value Function Decomposition for Cooperative Multi-agent Reinforcement Learning in StarCraft","date":"2022-08-15","arxiv_id":"2208.07298","repositories_listed":1,"syntology":null},{"url":"/paper/a-modular-framework-for-reinforcement","slug":"a-modular-framework-for-reinforcement","title":"A Modular Framework for Reinforcement Learning Optimal Execution","date":"2022-08-11","arxiv_id":"2208.06244","repositories_listed":1,"syntology":null},{"url":"/paper/robust-reinforcement-learning-using-offline","slug":"robust-reinforcement-learning-using-offline","title":"Robust Reinforcement Learning using Offline Data","date":"2022-08-10","arxiv_id":"2208.05129","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/robust-reinforcement-learning-using-offline#ran","syntology_url":"https://syntology.ai/paper/2208.05129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.05129"}},"official":{"repos":["zaiyan-x/RFQI"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/basis-for-intentions-efficient-inverse","slug":"basis-for-intentions-efficient-inverse","title":"Basis for Intentions: Efficient Inverse Reinforcement Learning using Past Experience","date":"2022-08-09","arxiv_id":"2208.04919","repositories_listed":1,"syntology":null},{"url":"/paper/from-scratch-to-sketch-deep-decoupled","slug":"from-scratch-to-sketch-deep-decoupled","title":"From Scratch to Sketch: Deep Decoupled Hierarchical Reinforcement Learning for Robotic Sketching Agent","date":"2022-08-09","arxiv_id":"2208.04833","repositories_listed":1,"syntology":null},{"url":"/paper/object-detection-with-deep-reinforcement","slug":"object-detection-with-deep-reinforcement","title":"Object Detection with Deep Reinforcement Learning","date":"2022-08-09","arxiv_id":"2208.04511","repositories_listed":1,"syntology":null},{"url":"/paper/socially-intelligent-genetic-agents-for-the","slug":"socially-intelligent-genetic-agents-for-the","title":"Socially Intelligent Genetic Agents for the Emergence of Explicit Norms","date":"2022-08-07","arxiv_id":"2208.03789","repositories_listed":1,"syntology":null},{"url":"/paper/a-cooperation-graph-approach-for-multiagent","slug":"a-cooperation-graph-approach-for-multiagent","title":"A Cooperation Graph Approach for Multiagent Sparse Reward Reinforcement Learning","date":"2022-08-05","arxiv_id":"2208.03002","repositories_listed":1,"syntology":null},{"url":"/paper/towards-augmented-microscopy-with","slug":"towards-augmented-microscopy-with","title":"Towards Augmented Microscopy with Reinforcement Learning-Enhanced Workflows","date":"2022-08-04","arxiv_id":"2208.02865","repositories_listed":1,"syntology":null},{"url":"/paper/mobility-aware-cooperative-caching-in","slug":"mobility-aware-cooperative-caching-in","title":"Mobility-Aware Cooperative Caching in Vehicular Edge Computing Based on Asynchronous Federated and Deep Reinforcement Learning","date":"2022-08-02","arxiv_id":"2208.01219","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-graph-reinforcement-learning-for","slug":"model-based-graph-reinforcement-learning-for","title":"Model-based graph reinforcement learning for inductive traffic signal control","date":"2022-08-01","arxiv_id":"2208.00659","repositories_listed":1,"syntology":null},{"url":"/paper/unified-automatic-control-of-vehicular","slug":"unified-automatic-control-of-vehicular","title":"Unified Automatic Control of Vehicular Systems with Reinforcement Learning","date":"2022-07-30","arxiv_id":"2208.00268","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-ucb-provably-efficient","slug":"contrastive-ucb-provably-efficient","title":"Contrastive UCB: Provably Efficient Contrastive Self-Supervised Learning in Online Reinforcement Learning","date":"2022-07-29","arxiv_id":"2207.14800","repositories_listed":1,"syntology":null},{"url":"/paper/cyclic-policy-distillation-sample-efficient","slug":"cyclic-policy-distillation-sample-efficient","title":"Cyclic Policy Distillation: Sample-Efficient Sim-to-Real Reinforcement Learning with Domain Randomization","date":"2022-07-29","arxiv_id":"2207.14561","repositories_listed":1,"syntology":null},{"url":"/paper/sampling-attacks-on-meta-reinforcement","slug":"sampling-attacks-on-meta-reinforcement","title":"Sampling Attacks on Meta Reinforcement Learning: A Minimax Formulation and Complexity Analysis","date":"2022-07-29","arxiv_id":"2208.00081","repositories_listed":1,"syntology":null},{"url":"/paper/post-processing-networks-method-for","slug":"post-processing-networks-method-for","title":"Post-processing Networks: Method for Optimizing Pipeline Task-oriented Dialogue Systems using Reinforcement Learning","date":"2022-07-25","arxiv_id":"2207.12185","repositories_listed":1,"syntology":null},{"url":"/paper/driver-dojo-a-benchmark-for-generalizable","slug":"driver-dojo-a-benchmark-for-generalizable","title":"Driver Dojo: A Benchmark for Generalizable Reinforcement Learning for Autonomous Driving","date":"2022-07-23","arxiv_id":"2207.11432","repositories_listed":1,"syntology":null},{"url":"/paper/robust-knowledge-adaptation-for-dynamic-graph","slug":"robust-knowledge-adaptation-for-dynamic-graph","title":"Robust Knowledge Adaptation for Dynamic Graph Neural Networks","date":"2022-07-22","arxiv_id":"2207.10839","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-implementation-of-a-reinforcement","slug":"on-the-implementation-of-a-reinforcement","title":"On the Implementation of a Reinforcement Learning-based Capacity Sharing Algorithm in O-RAN","date":"2022-07-21","arxiv_id":"2207.10390","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-energies-of-the","slug":"reinforcement-learning-for-energies-of-the","title":"Reinforcement learning for Energies of the future and carbon neutrality: a Challenge Design","date":"2022-07-21","arxiv_id":"2207.10330","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-market-making-1","slug":"deep-reinforcement-learning-for-market-making-1","title":"Deep Reinforcement Learning for Market Making Under a Hawkes Process-Based Limit Order Book Model","date":"2022-07-20","arxiv_id":"2207.09951","repositories_listed":1,"syntology":null},{"url":"/paper/generalizing-goal-conditioned-reinforcement","slug":"generalizing-goal-conditioned-reinforcement","title":"Generalizing Goal-Conditioned Reinforcement Learning with Variational Causal Reasoning","date":"2022-07-19","arxiv_id":"2207.09081","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalizing-goal-conditioned-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2207.09081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.09081"}},"official":{"repos":["gilgameshd/grader"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/magpie-automatically-tuning-static-parameters","slug":"magpie-automatically-tuning-static-parameters","title":"Magpie: Automatically Tuning Static Parameters for Distributed File Systems using Deep Reinforcement Learning","date":"2022-07-19","arxiv_id":"2207.09298","repositories_listed":1,"syntology":null},{"url":"/paper/a-meta-reinforcement-learning-algorithm-for","slug":"a-meta-reinforcement-learning-algorithm-for","title":"A Meta-Reinforcement Learning Algorithm for Causal Discovery","date":"2022-07-18","arxiv_id":"2207.08457","repositories_listed":1,"syntology":null},{"url":"/paper/active-exploration-for-inverse-reinforcement","slug":"active-exploration-for-inverse-reinforcement","title":"Active Exploration for Inverse Reinforcement Learning","date":"2022-07-18","arxiv_id":"2207.08645","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":3,"n_ran_checked":3,"n_instrument":7,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"10 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 7 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/active-exploration-for-inverse-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2207.08645","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.08645"}},"official":{"repos":["lasgroup/aceirl"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/asset-allocation-from-markowitz-to-deep","slug":"asset-allocation-from-markowitz-to-deep","title":"Asset Allocation: From Markowitz to Deep Reinforcement Learning","date":"2022-07-14","arxiv_id":"2208.07158","repositories_listed":1,"syntology":null},{"url":"/paper/a-general-contextualized-rewriting-framework","slug":"a-general-contextualized-rewriting-framework","title":"A General Contextualized Rewriting Framework for Text Summarization","date":"2022-07-13","arxiv_id":"2207.05948","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-assisted-recursive","slug":"reinforcement-learning-assisted-recursive","title":"Reinforcement Learning Assisted Recursive QAOA","date":"2022-07-13","arxiv_id":"2207.06294","repositories_listed":1,"syntology":null},{"url":"/paper/online-game-level-generation-from-music","slug":"online-game-level-generation-from-music","title":"Online Game Level Generation from Music","date":"2022-07-12","arxiv_id":"2207.05271","repositories_listed":1,"syntology":null},{"url":"/paper/reactive-exploration-to-cope-with-non","slug":"reactive-exploration-to-cope-with-non","title":"Reactive Exploration to Cope with Non-Stationarity in Lifelong Reinforcement Learning","date":"2022-07-12","arxiv_id":"2207.05742","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-disentanglement-of-representations","slug":"temporal-disentanglement-of-representations","title":"Temporal Disentanglement of Representations for Improved Generalisation in Reinforcement Learning","date":"2022-07-12","arxiv_id":"2207.05480","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/temporal-disentanglement-of-representations#ran","syntology_url":"https://syntology.ai/paper/2207.05480","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05480"}},"official":{"repos":["uoe-agents/ted"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/composuite-a-compositional-reinforcement","slug":"composuite-a-compositional-reinforcement","title":"CompoSuite: A Compositional Reinforcement Learning Benchmark","date":"2022-07-08","arxiv_id":"2207.04136","repositories_listed":1,"syntology":null},{"url":"/paper/interaction-pattern-disentangling-for-multi","slug":"interaction-pattern-disentangling-for-multi","title":"Interaction Pattern Disentangling for Multi-Agent Reinforcement Learning","date":"2022-07-08","arxiv_id":"2207.03902","repositories_listed":1,"syntology":null},{"url":"/paper/reinforced-lin-kernighan-helsgaun-algorithms","slug":"reinforced-lin-kernighan-helsgaun-algorithms","title":"Reinforced Lin-Kernighan-Helsgaun Algorithms for the Traveling Salesman Problems","date":"2022-07-08","arxiv_id":"2207.03876","repositories_listed":1,"syntology":null},{"url":"/paper/storehouse-a-reinforcement-learning","slug":"storehouse-a-reinforcement-learning","title":"Storehouse: a Reinforcement Learning Environment for Optimizing Warehouse Management","date":"2022-07-08","arxiv_id":"2207.03851","repositories_listed":1,"syntology":null},{"url":"/paper/robust-optimal-well-control-using-an-adaptive","slug":"robust-optimal-well-control-using-an-adaptive","title":"Robust optimal well control using an adaptive multi-grid reinforcement learning framework","date":"2022-07-07","arxiv_id":"2207.03253","repositories_listed":1,"syntology":null},{"url":"/paper/stochastic-optimal-well-control-in-subsurface","slug":"stochastic-optimal-well-control-in-subsurface","title":"Stochastic optimal well control in subsurface reservoirs using reinforcement learning","date":"2022-07-07","arxiv_id":"2207.03456","repositories_listed":1,"syntology":null},{"url":"/paper/implementing-reinforcement-learning-1","slug":"implementing-reinforcement-learning-1","title":"Implementing Reinforcement Learning Datacenter Congestion Control in NVIDIA NICs","date":"2022-07-05","arxiv_id":"2207.02295","repositories_listed":1,"syntology":null},{"url":"/paper/learning-task-embeddings-for-teamwork","slug":"learning-task-embeddings-for-teamwork","title":"Learning Task Embeddings for Teamwork Adaptation in Multi-Agent Reinforcement Learning","date":"2022-07-05","arxiv_id":"2207.02249","repositories_listed":1,"syntology":null},{"url":"/paper/robust-reinforcement-learning-in-continuous","slug":"robust-reinforcement-learning-in-continuous","title":"Robust Reinforcement Learning in Continuous Control Tasks with Uncertainty Set Regularization","date":"2022-07-05","arxiv_id":"2207.02016","repositories_listed":1,"syntology":null},{"url":"/paper/solving-the-traveling-salesperson-problem","slug":"solving-the-traveling-salesperson-problem","title":"Solving the Traveling Salesperson Problem with Precedence Constraints by Deep Reinforcement Learning","date":"2022-07-04","arxiv_id":"2207.01443","repositories_listed":1,"syntology":null},{"url":"/paper/stabilizing-off-policy-deep-reinforcement","slug":"stabilizing-off-policy-deep-reinforcement","title":"Stabilizing Off-Policy Deep Reinforcement Learning from Pixels","date":"2022-07-03","arxiv_id":"2207.00986","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/stabilizing-off-policy-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2207.00986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.00986"}},"official":{"repos":["aladoro/stabilizing-off-policy-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/usher-unbiased-sampling-for-hindsight","slug":"usher-unbiased-sampling-for-hindsight","title":"USHER: Unbiased Sampling for Hindsight Experience Replay","date":"2022-07-03","arxiv_id":"2207.01115","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/usher-unbiased-sampling-for-hindsight#ran","syntology_url":"https://syntology.ai/paper/2207.01115","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.01115"}},"official":null}}],"record_sha256":"cefe2115528ea3ef398e6e854c4dd8192422721e160d773bb54e93eda45955b4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}