{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/25","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":25,"pages_in_order":135,"rows_per_page":100,"rows":[2401,2500],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/24","next":"/task/reinforcement-learning-2/papers/26","papers":[{"url":"/paper/upside-down-reinforcement-learning-can","slug":"upside-down-reinforcement-learning-can","title":"Upside-Down Reinforcement Learning Can Diverge in Stochastic Environments With Episodic Resets","date":"2022-05-13","arxiv_id":"2205.06595","repositories_listed":1,"syntology":null},{"url":"/paper/a-state-distribution-matching-approach-to-non","slug":"a-state-distribution-matching-approach-to-non","title":"A State-Distribution Matching Approach to Non-Episodic Reinforcement Learning","date":"2022-05-11","arxiv_id":"2205.05212","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-state-distribution-matching-approach-to-non#ran","syntology_url":"https://syntology.ai/paper/2205.05212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.05212"}},"official":{"repos":["architsharma97/medal"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/intelligent-reflecting-surface-configurations","slug":"intelligent-reflecting-surface-configurations","title":"Intelligent Reflecting Surface Configurations for Smart Radio Using Deep Reinforcement Learning","date":"2022-05-11","arxiv_id":"2205.05269","repositories_listed":1,"syntology":null},{"url":"/paper/gamma-and-vega-hedging-using-deep","slug":"gamma-and-vega-hedging-using-deep","title":"Gamma and Vega Hedging Using Deep Distributional Reinforcement Learning","date":"2022-05-10","arxiv_id":"2205.05614","repositories_listed":1,"syntology":null},{"url":"/paper/state-encoders-in-reinforcement-learning-for","slug":"state-encoders-in-reinforcement-learning-for","title":"State Encoders in Reinforcement Learning for Recommendation: A Reproducibility Study","date":"2022-05-10","arxiv_id":"2205.04797","repositories_listed":1,"syntology":null},{"url":"/paper/vesnet-rl-simulation-based-reinforcement","slug":"vesnet-rl-simulation-based-reinforcement","title":"VesNet-RL: Simulation-based Reinforcement Learning for Real-World US Probe Navigation","date":"2022-05-10","arxiv_id":"2205.06676","repositories_listed":1,"syntology":null},{"url":"/paper/rlflow-optimising-neural-network-subgraph","slug":"rlflow-optimising-neural-network-subgraph","title":"RLFlow: Optimising Neural Network Subgraph Transformation with World Models","date":"2022-05-03","arxiv_id":"2205.01435","repositories_listed":1,"syntology":null},{"url":"/paper/cclf-a-contrastive-curiosity-driven-learning","slug":"cclf-a-contrastive-curiosity-driven-learning","title":"CCLF: A Contrastive-Curiosity-Driven Learning Framework for Sample-Efficient Reinforcement Learning","date":"2022-05-02","arxiv_id":"2205.00943","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/cclf-a-contrastive-curiosity-driven-learning#ran","syntology_url":"https://syntology.ai/paper/2205.00943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.00943"}},"official":{"repos":["csun001/cclf"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/large-neighborhood-search-based-on-neural","slug":"large-neighborhood-search-based-on-neural","title":"Large Neighborhood Search based on Neural Construction Heuristics","date":"2022-05-02","arxiv_id":"2205.00772","repositories_listed":1,"syntology":null},{"url":"/paper/ttopt-a-maximum-volume-quantized-tensor-train","slug":"ttopt-a-maximum-volume-quantized-tensor-train","title":"TTOpt: A Maximum Volume Quantized Tensor Train-based Optimization and its Application to Reinforcement Learning","date":"2022-04-30","arxiv_id":"2205.00293","repositories_listed":1,"syntology":{"n":17,"n_ran":14,"n_constructed":0,"n_ran_checked":6,"n_instrument":8,"n_unverified":3,"n_honours":6,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 6 honoured, 0 violated, 0 with no contract checked; 8 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ttopt-a-maximum-volume-quantized-tensor-train#ran","syntology_url":"https://syntology.ai/paper/2205.00293","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.00293"}},"official":{"repos":["andreichertkov/ttopt"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/cost-effective-mlaas-federation-a","slug":"cost-effective-mlaas-federation-a","title":"Cost Effective MLaaS Federation: A Combinatorial Reinforcement Learning Approach","date":"2022-04-29","arxiv_id":"2204.13971","repositories_listed":1,"syntology":null},{"url":"/paper/markov-abstractions-for-pac-reinforcement","slug":"markov-abstractions-for-pac-reinforcement","title":"Markov Abstractions for PAC Reinforcement Learning in Non-Markov Decision Processes","date":"2022-04-29","arxiv_id":"2205.01053","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/markov-abstractions-for-pac-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2205.01053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.01053"}},"official":{"repos":["whitemech/markov-abstractions-code-ijcai22"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/using-fuzzy-logic-to-learn-abstract-policies","slug":"using-fuzzy-logic-to-learn-abstract-policies","title":"Using Fuzzy Logic to Learn Abstract Policies in Large-Scale Multi-Agent Reinforcement Learning","date":"2022-04-27","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-for-11","slug":"multi-agent-reinforcement-learning-for-11","title":"Multi-Agent Reinforcement Learning for Traffic Signal Control through Universal Communication Method","date":"2022-04-26","arxiv_id":"2204.12190","repositories_listed":1,"syntology":null},{"url":"/paper/social-learning-spontaneously-emerges-by","slug":"social-learning-spontaneously-emerges-by","title":"Social learning spontaneously emerges by searching optimal heuristics with deep reinforcement learning","date":"2022-04-26","arxiv_id":"2204.12371","repositories_listed":1,"syntology":null},{"url":"/paper/toward-policy-explanations-for-multi-agent","slug":"toward-policy-explanations-for-multi-agent","title":"Toward Policy Explanations for Multi-Agent Reinforcement Learning","date":"2022-04-26","arxiv_id":"2204.12568","repositories_listed":1,"syntology":null},{"url":"/paper/hypernca-growing-developmental-networks-with","slug":"hypernca-growing-developmental-networks-with","title":"HyperNCA: Growing Developmental Networks with Neural Cellular Automata","date":"2022-04-25","arxiv_id":"2204.11674","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hypernca-growing-developmental-networks-with#ran","syntology_url":"https://syntology.ai/paper/2204.11674","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.11674"}},"official":null}},{"url":"/paper/multi-objective-pointer-network-for","slug":"multi-objective-pointer-network-for","title":"Multi-objective Pointer Network for Combinatorial Optimization","date":"2022-04-25","arxiv_id":"2204.11860","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-objective-pointer-network-for#ran","syntology_url":"https://syntology.ai/paper/2204.11860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.11860"}},"official":{"repos":["gaoly/mopn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/predicting-real-time-scientific-experiments","slug":"predicting-real-time-scientific-experiments","title":"Predicting Real-time Scientific Experiments Using Transformer models and Reinforcement Learning","date":"2022-04-25","arxiv_id":"2204.11718","repositories_listed":1,"syntology":null},{"url":"/paper/towards-evaluating-adaptivity-of-model-based","slug":"towards-evaluating-adaptivity-of-model-based","title":"Towards Evaluating Adaptivity of Model-Based Reinforcement Learning Methods","date":"2022-04-25","arxiv_id":"2204.11464","repositories_listed":1,"syntology":null},{"url":"/paper/reward-reports-for-reinforcement-learning","slug":"reward-reports-for-reinforcement-learning","title":"Reward Reports for Reinforcement Learning","date":"2022-04-22","arxiv_id":"2204.10817","repositories_listed":1,"syntology":null},{"url":"/paper/6gan-ipv6-multi-pattern-target-generation-via","slug":"6gan-ipv6-multi-pattern-target-generation-via","title":"6GAN: IPv6 Multi-Pattern Target Generation via Generative Adversarial Nets with Reinforcement Learning","date":"2022-04-21","arxiv_id":"2204.09839","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-gaussian-mixture-critic-in-off","slug":"revisiting-gaussian-mixture-critic-in-off","title":"Revisiting Gaussian mixture critics in off-policy reinforcement learning: a sample-based approach","date":"2022-04-21","arxiv_id":"2204.10256","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforcement-learning-based-volt-var","slug":"a-reinforcement-learning-based-volt-var","title":"A Reinforcement Learning-based Volt-VAR Control Dataset and Testing Environment","date":"2022-04-20","arxiv_id":"2204.09500","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-a-two-echelon","slug":"deep-reinforcement-learning-for-a-two-echelon","title":"Comparing Deep Reinforcement Learning Algorithms in Two-Echelon Supply Chains","date":"2022-04-20","arxiv_id":"2204.09603","repositories_listed":1,"syntology":null},{"url":"/paper/coptidice-offline-constrained-reinforcement-1","slug":"coptidice-offline-constrained-reinforcement-1","title":"COptiDICE: Offline Constrained Reinforcement Learning via Stationary Distribution Correction Estimation","date":"2022-04-19","arxiv_id":"2204.08957","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/coptidice-offline-constrained-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2204.08957","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.08957"}},"official":{"repos":["deepmind/constrained_optidice"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/fedkl-tackling-data-heterogeneity-in","slug":"fedkl-tackling-data-heterogeneity-in","title":"FedKL: Tackling Data Heterogeneity in Federated Reinforcement Learning by Penalizing KL Divergence","date":"2022-04-18","arxiv_id":"2204.08125","repositories_listed":1,"syntology":null},{"url":"/paper/safe-reinforcement-learning-using-black-box","slug":"safe-reinforcement-learning-using-black-box","title":"Safe Reinforcement Learning Using Black-Box Reachability Analysis","date":"2022-04-15","arxiv_id":"2204.07417","repositories_listed":1,"syntology":null},{"url":"/paper/can-question-rewriting-help-conversational","slug":"can-question-rewriting-help-conversational","title":"Can Question Rewriting Help Conversational Question Answering?","date":"2022-04-13","arxiv_id":"2204.06239","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-question-rewriting-help-conversational#ran","syntology_url":"https://syntology.ai/paper/2204.06239","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.06239"}},"official":{"repos":["hltchkust/cqr4cqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/gtlo-a-generalized-and-non-linear-multi","slug":"gtlo-a-generalized-and-non-linear-multi","title":"gTLO: A Generalized and Non-linear Multi-Objective Deep Reinforcement Learning Approach","date":"2022-04-11","arxiv_id":"2204.04988","repositories_listed":1,"syntology":null},{"url":"/paper/jorldy-a-fully-customizable-open-source","slug":"jorldy-a-fully-customizable-open-source","title":"JORLDY: a fully customizable open source framework for reinforcement learning","date":"2022-04-11","arxiv_id":"2204.04892","repositories_listed":1,"syntology":null},{"url":"/paper/confidence-estimation-transformer-for-long","slug":"confidence-estimation-transformer-for-long","title":"Confidence Estimation Transformer for Long-term Renewable Energy Forecasting in Reinforcement Learning-based Power Grid Dispatching","date":"2022-04-10","arxiv_id":"2204.04612","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-alignment-for-history-representation","slug":"temporal-alignment-for-history-representation","title":"Temporal Alignment for History Representation in Reinforcement Learning","date":"2022-04-07","arxiv_id":"2204.03525","repositories_listed":1,"syntology":null},{"url":"/paper/federated-reinforcement-learning-with","slug":"federated-reinforcement-learning-with","title":"Federated Reinforcement Learning with Environment Heterogeneity","date":"2022-04-06","arxiv_id":"2204.02634","repositories_listed":1,"syntology":null},{"url":"/paper/automating-reinforcement-learning-with","slug":"automating-reinforcement-learning-with","title":"Automating Reinforcement Learning with Example-based Resets","date":"2022-04-05","arxiv_id":"2204.02041","repositories_listed":1,"syntology":null},{"url":"/paper/jump-start-reinforcement-learning","slug":"jump-start-reinforcement-learning","title":"Jump-Start Reinforcement Learning","date":"2022-04-05","arxiv_id":"2204.02372","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-bid-long-term-multi-agent","slug":"learning-to-bid-long-term-multi-agent","title":"Learning to Bid Long-Term: Multi-Agent Reinforcement Learning with Long-Term and Sparse Reward in Repeated Auction Games","date":"2022-04-05","arxiv_id":"2204.02268","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-distributed-reinforcement","slug":"multi-agent-distributed-reinforcement","title":"Multi-Agent Distributed Reinforcement Learning for Making Decentralized Offloading Decisions","date":"2022-04-05","arxiv_id":"2204.02267","repositories_listed":1,"syntology":null},{"url":"/paper/disentangling-abstraction-from-statistical","slug":"disentangling-abstraction-from-statistical","title":"Disentangling Abstraction from Statistical Pattern Matching in Human and Machine Learning","date":"2022-04-04","arxiv_id":"2204.01437","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-agents-in-colonel","slug":"reinforcement-learning-agents-in-colonel","title":"Reinforcement Learning Agents in Colonel Blotto","date":"2022-04-04","arxiv_id":"2204.02785","repositories_listed":1,"syntology":null},{"url":"/paper/value-gradient-weighted-model-based-1","slug":"value-gradient-weighted-model-based-1","title":"Value Gradient weighted Model-Based Reinforcement Learning","date":"2022-04-04","arxiv_id":"2204.01464","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-guided-by-provable","slug":"reinforcement-learning-guided-by-provable","title":"Reinforcement Learning Guided by Provable Normative Compliance","date":"2022-03-30","arxiv_id":"2203.16275","repositories_listed":1,"syntology":null},{"url":"/paper/text-driven-video-acceleration-a-weakly","slug":"text-driven-video-acceleration-a-weakly","title":"Text-Driven Video Acceleration: A Weakly-Supervised Reinforcement Learning Method","date":"2022-03-29","arxiv_id":"2203.15778","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-risk-tendency-nano-drone-navigation","slug":"adaptive-risk-tendency-nano-drone-navigation","title":"Adaptive Risk-Tendency: Nano Drone Navigation in Cluttered Environments with Distributional Reinforcement Learning","date":"2022-03-28","arxiv_id":"2203.14749","repositories_listed":1,"syntology":null},{"url":"/paper/image-quality-assessment-for-machine-learning","slug":"image-quality-assessment-for-machine-learning","title":"Image quality assessment for machine learning tasks using meta-reinforcement learning","date":"2022-03-27","arxiv_id":"2203.14258","repositories_listed":1,"syntology":null},{"url":"/paper/an-optical-controlling-environment-and","slug":"an-optical-controlling-environment-and","title":"An Optical Control Environment for Benchmarking Reinforcement Learning Algorithms","date":"2022-03-23","arxiv_id":"2203.12114","repositories_listed":1,"syntology":null},{"url":"/paper/asynchronous-reinforcement-learning-for-real","slug":"asynchronous-reinforcement-learning-for-real","title":"Asynchronous Reinforcement Learning for Real-Time Control of Physical Robots","date":"2022-03-23","arxiv_id":"2203.12759","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/asynchronous-reinforcement-learning-for-real#ran","syntology_url":"https://syntology.ai/paper/2203.12759","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.12759"}},"official":{"repos":["yufengyuan/ur5_async_rl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/possibility-before-utility-learning-and-using-1","slug":"possibility-before-utility-learning-and-using-1","title":"Possibility Before Utility: Learning And Using Hierarchical Affordances","date":"2022-03-23","arxiv_id":"2203.12686","repositories_listed":1,"syntology":null},{"url":"/paper/is-vanilla-policy-gradient-overlooked","slug":"is-vanilla-policy-gradient-overlooked","title":"Is Vanilla Policy Gradient Overlooked? Analyzing Deep Reinforcement Learning for Hanabi","date":"2022-03-22","arxiv_id":"2203.11656","repositories_listed":1,"syntology":null},{"url":"/paper/long-short-term-memory-for-spatial-encoding","slug":"long-short-term-memory-for-spatial-encoding","title":"Long Short-Term Memory for Spatial Encoding in Multi-Agent Path Planning","date":"2022-03-21","arxiv_id":"2203.10823","repositories_listed":1,"syntology":null},{"url":"/paper/reccover-detecting-causal-confusion-for","slug":"reccover-detecting-causal-confusion-for","title":"ReCCoVER: Detecting Causal Confusion for Explainable Reinforcement Learning","date":"2022-03-21","arxiv_id":"2203.11211","repositories_listed":1,"syntology":null},{"url":"/paper/microracer-a-didactic-environment-for-deep","slug":"microracer-a-didactic-environment-for-deep","title":"MicroRacer: a didactic environment for Deep Reinforcement Learning","date":"2022-03-20","arxiv_id":"2203.10494","repositories_listed":1,"syntology":null},{"url":"/paper/perceiving-the-world-question-guided","slug":"perceiving-the-world-question-guided","title":"Perceiving the World: Question-guided Reinforcement Learning for Text-based Games","date":"2022-03-20","arxiv_id":"2204.09597","repositories_listed":1,"syntology":null},{"url":"/paper/quantum-multi-agent-reinforcement-learning","slug":"quantum-multi-agent-reinforcement-learning","title":"Quantum Multi-Agent Reinforcement Learning via Variational Quantum Circuit Design","date":"2022-03-20","arxiv_id":"2203.10443","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-automatic","slug":"reinforcement-learning-for-automatic","title":"Reinforcement learning for automatic quadrilateral mesh generation: a soft actor-critic approach","date":"2022-03-19","arxiv_id":"2203.11203","repositories_listed":1,"syntology":null},{"url":"/paper/teachable-reinforcement-learning-via-advice-1","slug":"teachable-reinforcement-learning-via-advice-1","title":"Teachable Reinforcement Learning via Advice Distillation","date":"2022-03-19","arxiv_id":"2203.11197","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/teachable-reinforcement-learning-via-advice-1#ran","syntology_url":"https://syntology.ai/paper/2203.11197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.11197"}},"official":{"repos":["rll-research/teachable"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/gac-a-deep-reinforcement-learning-model","slug":"gac-a-deep-reinforcement-learning-model","title":"GAC: A Deep Reinforcement Learning Model Toward User Incentivization in Unknown Social Networks","date":"2022-03-17","arxiv_id":"2203.09578","repositories_listed":1,"syntology":null},{"url":"/paper/semi-markov-offline-reinforcement-learning","slug":"semi-markov-offline-reinforcement-learning","title":"Semi-Markov Offline Reinforcement Learning for Healthcare","date":"2022-03-17","arxiv_id":"2203.09365","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/semi-markov-offline-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2203.09365","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.09365"}},"official":{"repos":["mary-wu/smdp"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/coach-assisted-multi-agent-reinforcement","slug":"coach-assisted-multi-agent-reinforcement","title":"Coach-assisted Multi-Agent Reinforcement Learning Framework for Unexpected Crashed Agents","date":"2022-03-16","arxiv_id":"2203.08454","repositories_listed":1,"syntology":null},{"url":"/paper/copa-certifying-robust-policies-for-offline-1","slug":"copa-certifying-robust-policies-for-offline-1","title":"COPA: Certifying Robust Policies for Offline Reinforcement Learning against Poisoning Attacks","date":"2022-03-16","arxiv_id":"2203.08398","repositories_listed":1,"syntology":null},{"url":"/paper/ctds-centralized-teacher-with-decentralized","slug":"ctds-centralized-teacher-with-decentralized","title":"CTDS: Centralized Teacher with Decentralized Student for Multi-Agent Reinforcement Learning","date":"2022-03-16","arxiv_id":"2203.08412","repositories_listed":1,"syntology":null},{"url":"/paper/latent-variable-advantage-weighted-policy","slug":"latent-variable-advantage-weighted-policy","title":"Latent-Variable Advantage-Weighted Policy Optimization for Offline RL","date":"2022-03-16","arxiv_id":"2203.08949","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/latent-variable-advantage-weighted-policy#ran","syntology_url":"https://syntology.ai/paper/2203.08949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08949"}},"official":null}},{"url":"/paper/pmic-improving-multi-agent-reinforcement-1","slug":"pmic-improving-multi-agent-reinforcement-1","title":"PMIC: Improving Multi-Agent Reinforcement Learning with Progressive Mutual Information Collaboration","date":"2022-03-16","arxiv_id":"2203.08553","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pmic-improving-multi-agent-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2203.08553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08553"}},"official":{"repos":["yeshenpy/pmic"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/zipfian-environments-for-reinforcement","slug":"zipfian-environments-for-reinforcement","title":"Zipfian environments for Reinforcement Learning","date":"2022-03-15","arxiv_id":"2203.08222","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/zipfian-environments-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2203.08222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08222"}},"official":{"repos":["deepmind/zipfian_environments"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/l2explorer-a-lifelong-reinforcement-learning","slug":"l2explorer-a-lifelong-reinforcement-learning","title":"L2Explorer: A Lifelong Reinforcement Learning Assessment Environment","date":"2022-03-14","arxiv_id":"2203.07454","repositories_listed":1,"syntology":null},{"url":"/paper/orchestrated-value-mapping-for-reinforcement-1","slug":"orchestrated-value-mapping-for-reinforcement-1","title":"Orchestrated Value Mapping for Reinforcement Learning","date":"2022-03-14","arxiv_id":"2203.07171","repositories_listed":1,"syntology":null},{"url":"/paper/near-optimal-deep-reinforcement-learning","slug":"near-optimal-deep-reinforcement-learning","title":"Near-optimal Deep Reinforcement Learning Policies from Data for Zone Temperature Control","date":"2022-03-10","arxiv_id":"2203.05434","repositories_listed":1,"syntology":null},{"url":"/paper/neuro-symbolic-natural-logic-with","slug":"neuro-symbolic-natural-logic-with","title":"Neuro-symbolic Natural Logic with Introspective Revision for Natural Language Inference","date":"2022-03-09","arxiv_id":"2203.04857","repositories_listed":1,"syntology":null},{"url":"/paper/sage-generating-symbolic-goals-for-myopic","slug":"sage-generating-symbolic-goals-for-myopic","title":"SAGE: Generating Symbolic Goals for Myopic Models in Deep Reinforcement Learning","date":"2022-03-09","arxiv_id":"2203.05079","repositories_listed":1,"syntology":null},{"url":"/paper/curriculum-based-reinforcement-learning-for","slug":"curriculum-based-reinforcement-learning-for","title":"Curriculum-based Reinforcement Learning for Distribution System Critical Load Restoration","date":"2022-03-08","arxiv_id":"2203.04166","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-entity-1","slug":"deep-reinforcement-learning-for-entity-1","title":"Deep Reinforcement Learning for Entity Alignment","date":"2022-03-07","arxiv_id":"2203.03315","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-for-entity-1#ran","syntology_url":"https://syntology.ai/paper/2203.03315","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.03315"}},"official":{"repos":["guolingbing/rlea"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/influencing-long-term-behavior-in-multiagent","slug":"influencing-long-term-behavior-in-multiagent","title":"Influencing Long-Term Behavior in Multiagent Reinforcement Learning","date":"2022-03-07","arxiv_id":"2203.03535","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":3,"n_ran_checked":4,"n_instrument":6,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"10 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/influencing-long-term-behavior-in-multiagent#ran","syntology_url":"https://syntology.ai/paper/2203.03535","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.03535"}},"official":{"repos":["dkkim93/further"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/on-credit-assignment-in-hierarchical","slug":"on-credit-assignment-in-hierarchical","title":"On Credit Assignment in Hierarchical Reinforcement Learning","date":"2022-03-07","arxiv_id":"2203.03292","repositories_listed":1,"syntology":null},{"url":"/paper/reliably-re-acting-to-partner-s-actions-with","slug":"reliably-re-acting-to-partner-s-actions-with","title":"Reliably Re-Acting to Partner's Actions with the Social Intrinsic Motivation of Transfer Empowerment","date":"2022-03-07","arxiv_id":"2203.03355","repositories_listed":1,"syntology":null},{"url":"/paper/on-practical-reinforcement-learning-provable","slug":"on-practical-reinforcement-learning-provable","title":"On Practical Reinforcement Learning: Provable Robustness, Scalability, and Statistical Efficiency","date":"2022-03-03","arxiv_id":"2203.01758","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-in-possibly","slug":"reinforcement-learning-in-possibly","title":"Testing Stationarity and Change Point Detection in Reinforcement Learning","date":"2022-03-03","arxiv_id":"2203.01707","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-offline-reinforcement-learning","slug":"a-survey-on-offline-reinforcement-learning","title":"A Survey on Offline Reinforcement Learning: Taxonomy, Review, and Open Problems","date":"2022-03-02","arxiv_id":"2203.01387","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-survey-on-offline-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2203.01387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.01387"}},"official":{"repos":["larocs/offline-rl-suvey"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/andes-gym-a-versatile-environment-for-deep","slug":"andes-gym-a-versatile-environment-for-deep","title":"Andes_gym: A Versatile Environment for Deep Reinforcement Learning in Power Systems","date":"2022-03-02","arxiv_id":"2203.01292","repositories_listed":1,"syntology":null},{"url":"/paper/combining-reinforcement-learning-and-optimal","slug":"combining-reinforcement-learning-and-optimal","title":"Combining Reinforcement Learning and Optimal Transport for the Traveling Salesman Problem","date":"2022-03-02","arxiv_id":"2203.00903","repositories_listed":1,"syntology":null},{"url":"/paper/integrating-contrastive-learning-with-dynamic","slug":"integrating-contrastive-learning-with-dynamic","title":"Integrating Contrastive Learning with Dynamic Models for Reinforcement Learning from Images","date":"2022-03-02","arxiv_id":"2203.01810","repositories_listed":1,"syntology":null},{"url":"/paper/ai-planning-annotation-for-sample-efficient","slug":"ai-planning-annotation-for-sample-efficient","title":"Hierarchical Reinforcement Learning with AI Planning Models","date":"2022-03-01","arxiv_id":"2203.00669","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-generalization-of-representations-in","slug":"on-the-generalization-of-representations-in","title":"On the Generalization of Representations in Reinforcement Learning","date":"2022-03-01","arxiv_id":"2203.00543","repositories_listed":1,"syntology":null},{"url":"/paper/avalanche-rl-a-continual-reinforcement","slug":"avalanche-rl-a-continual-reinforcement","title":"Avalanche RL: a Continual Reinforcement Learning Library","date":"2022-02-28","arxiv_id":"2202.13657","repositories_listed":1,"syntology":null},{"url":"/paper/monkey-business-reinforcement-learning-meets","slug":"monkey-business-reinforcement-learning-meets","title":"Monkey Business: Reinforcement learning meets neighborhood search for Virtual Network Embedding","date":"2022-02-28","arxiv_id":"2202.13706","repositories_listed":1,"syntology":null},{"url":"/paper/probing-the-robustness-of-trained-metrics-for-1","slug":"probing-the-robustness-of-trained-metrics-for-1","title":"Probing the Robustness of Trained Metrics for Conversational Dialogue Systems","date":"2022-02-28","arxiv_id":"2202.13887","repositories_listed":1,"syntology":null},{"url":"/paper/rl-pgo-reinforcement-learning-based-planar","slug":"rl-pgo-reinforcement-learning-based-planar","title":"RL-PGO: Reinforcement Learning-based Planar Pose-Graph Optimization","date":"2022-02-26","arxiv_id":"2202.13221","repositories_listed":1,"syntology":null},{"url":"/paper/statistically-efficient-advantage-learning","slug":"statistically-efficient-advantage-learning","title":"Statistically Efficient Advantage Learning for Offline Reinforcement Learning in Infinite Horizons","date":"2022-02-26","arxiv_id":"2202.13163","repositories_listed":1,"syntology":null},{"url":"/paper/building-a-3-player-mahjong-ai-using-deep","slug":"building-a-3-player-mahjong-ai-using-deep","title":"Building a 3-Player Mahjong AI using Deep Reinforcement Learning","date":"2022-02-25","arxiv_id":"2202.12847","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-with-sticky-mittens-reinforcement","slug":"exploring-with-sticky-mittens-reinforcement","title":"Exploring with Sticky Mittens: Reinforcement Learning with Expert Interventions via Option Templates","date":"2022-02-25","arxiv_id":"2202.12967","repositories_listed":1,"syntology":null},{"url":"/paper/all-you-need-is-supervised-learning-from","slug":"all-you-need-is-supervised-learning-from","title":"All You Need Is Supervised Learning: From Imitation Learning to Meta-RL With Upside Down RL","date":"2022-02-24","arxiv_id":"2202.11960","repositories_listed":1,"syntology":null},{"url":"/paper/learning-transferable-reward-for-query-object-1","slug":"learning-transferable-reward-for-query-object-1","title":"Learning Transferable Reward for Query Object Localization with Policy Adaptation","date":"2022-02-24","arxiv_id":"2202.12403","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-transferable-reward-for-query-object-1#ran","syntology_url":"https://syntology.ai/paper/2202.12403","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.12403"}},"official":{"repos":["litingfeng/localization-by-ordembed"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/blockchain-framework-for-artificial","slug":"blockchain-framework-for-artificial","title":"Blockchain Framework for Artificial Intelligence Computation","date":"2022-02-23","arxiv_id":"2202.11264","repositories_listed":1,"syntology":null},{"url":"/paper/pessimistic-bootstrapping-for-uncertainty-1","slug":"pessimistic-bootstrapping-for-uncertainty-1","title":"Pessimistic Bootstrapping for Uncertainty-Driven Offline Reinforcement Learning","date":"2022-02-23","arxiv_id":"2202.11566","repositories_listed":1,"syntology":null},{"url":"/paper/using-deep-reinforcement-learning-with","slug":"using-deep-reinforcement-learning-with","title":"Using Deep Reinforcement Learning with Automatic Curriculum Learning for Mapless Navigation in Intralogistics","date":"2022-02-23","arxiv_id":"2202.11512","repositories_listed":1,"syntology":null},{"url":"/paper/a-comparative-study-of-deep-reinforcement","slug":"a-comparative-study-of-deep-reinforcement","title":"A Comparative Study of Deep Reinforcement Learning-based Transferable Energy Management Strategies for Hybrid Electric Vehicles","date":"2022-02-22","arxiv_id":"2202.11514","repositories_listed":1,"syntology":null},{"url":"/paper/shaping-advice-in-deep-reinforcement-learning","slug":"shaping-advice-in-deep-reinforcement-learning","title":"Shaping Advice in Deep Reinforcement Learning","date":"2022-02-19","arxiv_id":"2202.09489","repositories_listed":1,"syntology":null},{"url":"/paper/darl1n-distributed-multi-agent-reinforcement","slug":"darl1n-distributed-multi-agent-reinforcement","title":"Distributed Multi-Agent Reinforcement Learning with One-hop Neighbors and Compute Straggler Mitigation","date":"2022-02-18","arxiv_id":"2202.09019","repositories_listed":1,"syntology":null},{"url":"/paper/cadre-a-cascade-deep-reinforcement-learning","slug":"cadre-a-cascade-deep-reinforcement-learning","title":"CADRE: A Cascade Deep Reinforcement Learning Framework for Vision-based Autonomous Urban Driving","date":"2022-02-17","arxiv_id":"2202.08557","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":7,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 7 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cadre-a-cascade-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2202.08557","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.08557"}},"official":{"repos":["BIT-MCS/Cadre"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":7,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-intrinsic-exploration-with-language","slug":"improving-intrinsic-exploration-with-language","title":"Improving Intrinsic Exploration with Language Abstractions","date":"2022-02-17","arxiv_id":"2202.08938","repositories_listed":1,"syntology":null},{"url":"/paper/vrl3-a-data-driven-framework-for-visual-deep","slug":"vrl3-a-data-driven-framework-for-visual-deep","title":"VRL3: A Data-Driven Framework for Visual Deep Reinforcement Learning","date":"2022-02-17","arxiv_id":"2202.10324","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":4,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/vrl3-a-data-driven-framework-for-visual-deep#ran","syntology_url":"https://syntology.ai/paper/2202.10324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.10324"}},"official":{"repos":["facebookresearch/drqv2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}}],"record_sha256":"e7c553cea561dde023e9294c12f080d1122dfbc3ef99fbc15276111341cf64e9","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}