{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/26","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":26,"pages_in_order":132,"rows_per_page":100,"rows":[2501,2600],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/25","next":"/task/reinforcement-learning/papers/27","papers":[{"url":"/paper/autophoto-aesthetic-photo-capture-using","slug":"autophoto-aesthetic-photo-capture-using","title":"AutoPhoto: Aesthetic Photo Capture using Reinforcement Learning","date":"2021-09-21","arxiv_id":"2109.09923","repositories_listed":1,"syntology":null},{"url":"/paper/deep-policies-for-online-bipartite-matching-a","slug":"deep-policies-for-online-bipartite-matching-a","title":"Deep Policies for Online Bipartite Matching: A Reinforcement Learning Approach","date":"2021-09-21","arxiv_id":"2109.10380","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/deep-policies-for-online-bipartite-matching-a#ran","syntology_url":"https://syntology.ai/paper/2109.10380","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.10380"}},"official":{"repos":["lyeskhalil/corl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/generalization-in-text-based-games-via","slug":"generalization-in-text-based-games-via","title":"Generalization in Text-based Games via Hierarchical Reinforcement Learning","date":"2021-09-21","arxiv_id":"2109.09968","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalization-in-text-based-games-via#ran","syntology_url":"https://syntology.ai/paper/2109.09968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.09968"}},"official":{"repos":["yunqiuxu/h-kga"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hindsight-foresight-relabeling-for-meta","slug":"hindsight-foresight-relabeling-for-meta","title":"Hindsight Foresight Relabeling for Meta-Reinforcement Learning","date":"2021-09-18","arxiv_id":"2109.09031","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hindsight-foresight-relabeling-for-meta#ran","syntology_url":"https://syntology.ai/paper/2109.09031","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.09031"}},"official":{"repos":["michaelwan11/hfr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/back-to-basics-deep-reinforcement-learning-in","slug":"back-to-basics-deep-reinforcement-learning-in","title":"Back to Basics: Deep Reinforcement Learning in Traffic Signal Control","date":"2021-09-15","arxiv_id":"2109.07180","repositories_listed":1,"syntology":null},{"url":"/paper/dcur-data-curriculum-for-teaching-via-samples","slug":"dcur-data-curriculum-for-teaching-via-samples","title":"DCUR: Data Curriculum for Teaching via Samples with Reinforcement Learning","date":"2021-09-15","arxiv_id":"2109.07380","repositories_listed":1,"syntology":null},{"url":"/paper/estimation-of-warfarin-dosage-with","slug":"estimation-of-warfarin-dosage-with","title":"Estimation of Warfarin Dosage with Reinforcement Learning","date":"2021-09-15","arxiv_id":"2109.07564","repositories_listed":1,"syntology":null},{"url":"/paper/few-shot-quality-diversity-optimisation","slug":"few-shot-quality-diversity-optimisation","title":"Few-shot Quality-Diversity Optimization","date":"2021-09-14","arxiv_id":"2109.06826","repositories_listed":1,"syntology":null},{"url":"/paper/gradient-imitation-reinforcement-learning-for","slug":"gradient-imitation-reinforcement-learning-for","title":"Gradient Imitation Reinforcement Learning for Low Resource Relation Extraction","date":"2021-09-14","arxiv_id":"2109.06415","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gradient-imitation-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2109.06415","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.06415"}},"official":{"repos":["thu-bpm/gradlre"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-evolutionary","slug":"reinforcement-learning-with-evolutionary","title":"Reinforcement Learning with Evolutionary Trajectory Generator: A General Approach for Quadrupedal Locomotion","date":"2021-09-14","arxiv_id":"2109.06409","repositories_listed":1,"syntology":null},{"url":"/paper/wavecorr-correlation-savvy-deep-reinforcement","slug":"wavecorr-correlation-savvy-deep-reinforcement","title":"WaveCorr: Correlation-savvy Deep Reinforcement Learning for Portfolio Management","date":"2021-09-14","arxiv_id":"2109.07005","repositories_listed":1,"syntology":null},{"url":"/paper/direct-random-search-for-fine-tuning-of-deep","slug":"direct-random-search-for-fine-tuning-of-deep","title":"Direct Random Search for Fine Tuning of Deep Reinforcement Learning Policies","date":"2021-09-12","arxiv_id":"2109.05604","repositories_listed":1,"syntology":null},{"url":"/paper/opirl-sample-efficient-off-policy-inverse","slug":"opirl-sample-efficient-off-policy-inverse","title":"OPIRL: Sample Efficient Off-Policy Inverse Reinforcement Learning via Distribution Matching","date":"2021-09-09","arxiv_id":"2109.04307","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/opirl-sample-efficient-off-policy-inverse#ran","syntology_url":"https://syntology.ai/paper/2109.04307","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.04307"}},"official":{"repos":["sff1019/opirl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/timetraveler-reinforcement-learning-for","slug":"timetraveler-reinforcement-learning-for","title":"TimeTraveler: Reinforcement Learning for Temporal Knowledge Graph Forecasting","date":"2021-09-09","arxiv_id":"2109.04101","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/timetraveler-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2109.04101","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.04101"}},"official":{"repos":["jhl-hust/titer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/powergym-a-reinforcement-learning-environment","slug":"powergym-a-reinforcement-learning-environment","title":"PowerGym: A Reinforcement Learning Environment for Volt-Var Control in Power Distribution Systems","date":"2021-09-08","arxiv_id":"2109.03970","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/powergym-a-reinforcement-learning-environment#ran","syntology_url":"https://syntology.ai/paper/2109.03970","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.03970"}},"official":{"repos":["siemens/powergym"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/optimizing-quantum-variational-circuits-with","slug":"optimizing-quantum-variational-circuits-with","title":"Optimizing Quantum Variational Circuits with Deep Reinforcement Learning","date":"2021-09-07","arxiv_id":"2109.03188","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimizing-quantum-variational-circuits-with#ran","syntology_url":"https://syntology.ai/paper/2109.03188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.03188"}},"official":{"repos":["lockwo/rl_qvc_opt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/temporal-aware-deep-reinforcement-learning","slug":"temporal-aware-deep-reinforcement-learning","title":"Temporal Shift Reinforcement Learning","date":"2021-09-05","arxiv_id":"2109.02145","repositories_listed":1,"syntology":null},{"url":"/paper/catastrophic-interference-in-reinforcement","slug":"catastrophic-interference-in-reinforcement","title":"Catastrophic Interference in Reinforcement Learning: A Solution Based on Context Division and Knowledge Distillation","date":"2021-09-01","arxiv_id":"2109.00525","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/catastrophic-interference-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2109.00525","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.00525"}},"official":{"repos":["sweety-dm/interference-aware-deep-q-learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/regen-reinforcement-learning-for-text-and","slug":"regen-reinforcement-learning-for-text-and","title":"ReGen: Reinforcement Learning for Text and Knowledge Base Generation using Pretrained Language Models","date":"2021-08-27","arxiv_id":"2108.12472","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regen-reinforcement-learning-for-text-and#ran","syntology_url":"https://syntology.ai/paper/2108.12472","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.12472"}},"official":{"repos":["IBM/regen"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-based-condition","slug":"reinforcement-learning-based-condition","title":"Reinforcement Learning based Condition-oriented Maintenance Scheduling for Flow Line Systems","date":"2021-08-27","arxiv_id":"2108.12298","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-powered-semantic","slug":"reinforcement-learning-powered-semantic","title":"Reinforcement Learning-powered Semantic Communication via Semantic Similarity","date":"2021-08-27","arxiv_id":"2108.12121","repositories_listed":1,"syntology":null},{"url":"/paper/robust-risk-aware-reinforcement-learning","slug":"robust-risk-aware-reinforcement-learning","title":"Robust Risk-Aware Reinforcement Learning","date":"2021-08-23","arxiv_id":"2108.10403","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-state-augmentation-methods-for","slug":"revisiting-state-augmentation-methods-for","title":"Revisiting State Augmentation methods for Reinforcement Learning with Stochastic Delays","date":"2021-08-17","arxiv_id":"2108.07555","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/revisiting-state-augmentation-methods-for#ran","syntology_url":"https://syntology.ai/paper/2108.07555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.07555"}},"official":{"repos":["baranwa2/delayresolvedrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/aspect-sentiment-triplet-extraction-using","slug":"aspect-sentiment-triplet-extraction-using","title":"Aspect Sentiment Triplet Extraction Using Reinforcement Learning","date":"2021-08-13","arxiv_id":"2108.06107","repositories_listed":1,"syntology":null},{"url":"/paper/fairness-through-counterfactual-utilities","slug":"fairness-through-counterfactual-utilities","title":"Fairness Through Counterfactual Utilities","date":"2021-08-11","arxiv_id":"2108.05315","repositories_listed":1,"syntology":null},{"url":"/paper/gap-dependent-unsupervised-exploration-for","slug":"gap-dependent-unsupervised-exploration-for","title":"Gap-Dependent Unsupervised Exploration for Reinforcement Learning","date":"2021-08-11","arxiv_id":"2108.05439","repositories_listed":1,"syntology":null},{"url":"/paper/imitation-learning-by-reinforcement-learning","slug":"imitation-learning-by-reinforcement-learning","title":"Imitation Learning by Reinforcement Learning","date":"2021-08-10","arxiv_id":"2108.04763","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imitation-learning-by-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2108.04763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.04763"}},"official":{"repos":["spotify-research/il-by-rl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-reinforcement-learning-in-broad-and-non","slug":"meta-reinforcement-learning-in-broad-and-non","title":"Meta-Reinforcement Learning in Broad and Non-Parametric Environments","date":"2021-08-08","arxiv_id":"2108.03718","repositories_listed":1,"syntology":null},{"url":"/paper/what-matters-in-learning-from-offline-human","slug":"what-matters-in-learning-from-offline-human","title":"What Matters in Learning from Offline Human Demonstrations for Robot Manipulation","date":"2021-08-06","arxiv_id":"2108.03298","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/what-matters-in-learning-from-offline-human#ran","syntology_url":"https://syntology.ai/paper/2108.03298","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.03298"}},"official":null}},{"url":"/paper/an-encoder-decoder-based-audio-captioning","slug":"an-encoder-decoder-based-audio-captioning","title":"An Encoder-Decoder Based Audio Captioning System With Transfer and Reinforcement Learning","date":"2021-08-05","arxiv_id":"2108.02752","repositories_listed":1,"syntology":null},{"url":"/paper/a-pragmatic-look-at-deep-imitation-learning","slug":"a-pragmatic-look-at-deep-imitation-learning","title":"A Pragmatic Look at Deep Imitation Learning","date":"2021-08-04","arxiv_id":"2108.01867","repositories_listed":1,"syntology":null},{"url":"/paper/learning-task-agnostic-skills-with-data","slug":"learning-task-agnostic-skills-with-data","title":"Learning Task Agnostic Skills with Data-driven Guidance","date":"2021-08-04","arxiv_id":"2108.01869","repositories_listed":1,"syntology":null},{"url":"/paper/a-scalable-federated-multi-agent-architecture","slug":"a-scalable-federated-multi-agent-architecture","title":"Scalable Multi-agent Reinforcement Learning Algorithm for Wireless Networks","date":"2021-08-01","arxiv_id":"2108.00506","repositories_listed":1,"syntology":null},{"url":"/paper/how-helpful-is-inverse-reinforcement-learning","slug":"how-helpful-is-inverse-reinforcement-learning","title":"How Helpful is Inverse Reinforcement Learning for Table-to-Text Generation?","date":"2021-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/phrase-level-action-reinforcement-learning","slug":"phrase-level-action-reinforcement-learning","title":"Phrase-Level Action Reinforcement Learning for Neural Dialog Response Generation","date":"2021-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/rltutor-reinforcement-learning-based-adaptive","slug":"rltutor-reinforcement-learning-based-adaptive","title":"RLTutor: Reinforcement Learning Based Adaptive Tutoring System by Modeling Virtual Student with Fewer Interactions","date":"2021-07-31","arxiv_id":"2108.00268","repositories_listed":1,"syntology":null},{"url":"/paper/sequence-adaptation-via-reinforcement","slug":"sequence-adaptation-via-reinforcement","title":"Sequence Adaptation via Reinforcement Learning in Recommender Systems","date":"2021-07-31","arxiv_id":"2108.01442","repositories_listed":1,"syntology":null},{"url":"/paper/strategically-efficient-exploration-in","slug":"strategically-efficient-exploration-in","title":"Strategically Efficient Exploration in Competitive Multi-agent Reinforcement Learning","date":"2021-07-30","arxiv_id":"2107.14698","repositories_listed":1,"syntology":null},{"url":"/paper/tianshou-a-highly-modularized-deep","slug":"tianshou-a-highly-modularized-deep","title":"Tianshou: a Highly Modularized Deep Reinforcement Learning Library","date":"2021-07-29","arxiv_id":"2107.14171","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tianshou-a-highly-modularized-deep#ran","syntology_url":"https://syntology.ai/paper/2107.14171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.14171"}},"official":{"repos":["thu-ml/tianshou"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/accelerating-quadratic-optimization-with","slug":"accelerating-quadratic-optimization-with","title":"Accelerating Quadratic Optimization with Reinforcement Learning","date":"2021-07-22","arxiv_id":"2107.10847","repositories_listed":1,"syntology":null},{"url":"/paper/demonstration-guided-reinforcement-learning","slug":"demonstration-guided-reinforcement-learning","title":"Demonstration-Guided Reinforcement Learning with Learned Skills","date":"2021-07-21","arxiv_id":"2107.10253","repositories_listed":1,"syntology":null},{"url":"/paper/toward-collaborative-reinforcement-learning","slug":"toward-collaborative-reinforcement-learning","title":"Toward Collaborative Reinforcement Learning Agents that Communicate Through Text-Based Natural Language","date":"2021-07-20","arxiv_id":"2107.09356","repositories_listed":1,"syntology":null},{"url":"/paper/a-sustainable-ecosystem-through-emergent","slug":"a-sustainable-ecosystem-through-emergent","title":"A Sustainable Ecosystem through Emergent Cooperation in Multi-Agent Reinforcement Learning","date":"2021-07-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/decoupling-exploration-and-exploitation-in","slug":"decoupling-exploration-and-exploitation-in","title":"Decoupled Reinforcement Learning to Stabilise Intrinsically-Motivated Exploration","date":"2021-07-19","arxiv_id":"2107.08966","repositories_listed":1,"syntology":null},{"url":"/paper/vision-based-autonomous-car-racing-using-deep","slug":"vision-based-autonomous-car-racing-using-deep","title":"Vision-Based Autonomous Car Racing Using Deep Imitative Reinforcement Learning","date":"2021-07-18","arxiv_id":"2107.08325","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforcement-learning-environment-for-2","slug":"a-reinforcement-learning-environment-for-2","title":"A Reinforcement Learning Environment for Mathematical Reasoning via Program Synthesis","date":"2021-07-15","arxiv_id":"2107.07373","repositories_listed":1,"syntology":null},{"url":"/paper/deep-adaptive-multi-intention-inverse","slug":"deep-adaptive-multi-intention-inverse","title":"Deep Adaptive Multi-Intention Inverse Reinforcement Learning","date":"2021-07-14","arxiv_id":"2107.06692","repositories_listed":1,"syntology":null},{"url":"/paper/safer-reinforcement-learning-through","slug":"safer-reinforcement-learning-through","title":"Safer Reinforcement Learning through Transferable Instinct Networks","date":"2021-07-14","arxiv_id":"2107.06686","repositories_listed":1,"syntology":null},{"url":"/paper/carle-s-game-an-open-ended-challenge-in","slug":"carle-s-game-an-open-ended-challenge-in","title":"Carle's Game: An Open-Ended Challenge in Exploratory Machine Creativity","date":"2021-07-13","arxiv_id":"2107.05786","repositories_listed":1,"syntology":null},{"url":"/paper/rellie-deep-reinforcement-learning-for","slug":"rellie-deep-reinforcement-learning-for","title":"ReLLIE: Deep Reinforcement Learning for Customized Low-Light Image Enhancement","date":"2021-07-13","arxiv_id":"2107.05830","repositories_listed":1,"syntology":null},{"url":"/paper/shortest-path-constrained-reinforcement-1","slug":"shortest-path-constrained-reinforcement-1","title":"Shortest-Path Constrained Reinforcement Learning for Sparse Reward Tasks","date":"2021-07-13","arxiv_id":"2107.06405","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shortest-path-constrained-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2107.06405","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.06405"}},"official":{"repos":["srsohn/shortest-path-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/behavior-constraining-in-weight-space-for","slug":"behavior-constraining-in-weight-space-for","title":"Behavior Constraining in Weight Space for Offline Reinforcement Learning","date":"2021-07-12","arxiv_id":"2107.05479","repositories_listed":1,"syntology":null},{"url":"/paper/conservative-offline-distributional","slug":"conservative-offline-distributional","title":"Conservative Offline Distributional Reinforcement Learning","date":"2021-07-12","arxiv_id":"2107.06106","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/conservative-offline-distributional#ran","syntology_url":"https://syntology.ai/paper/2107.06106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.06106"}},"official":{"repos":["JasonMa2016/CODAC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/explore-and-control-with-adversarial-surprise","slug":"explore-and-control-with-adversarial-surprise","title":"Explore and Control with Adversarial Surprise","date":"2021-07-12","arxiv_id":"2107.07394","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/explore-and-control-with-adversarial-surprise#ran","syntology_url":"https://syntology.ai/paper/2107.07394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.07394"}},"official":null}},{"url":"/paper/modeling-explicit-concerning-states-for","slug":"modeling-explicit-concerning-states-for","title":"Modeling Explicit Concerning States for Reinforcement Learning in Visual Dialogue","date":"2021-07-12","arxiv_id":"2107.05250","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/modeling-explicit-concerning-states-for#ran","syntology_url":"https://syntology.ai/paper/2107.05250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.05250"}},"official":{"repos":["zipengxuc/ecs-visdial-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/backprop-free-reinforcement-learning-with","slug":"backprop-free-reinforcement-learning-with","title":"Backprop-Free Reinforcement Learning with Active Neural Generative Coding","date":"2021-07-10","arxiv_id":"2107.07046","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/backprop-free-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2107.07046","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.07046"}},"official":{"repos":["ago109/active-neural-generative-coding"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-meta-reinforcement-learning-with-1","slug":"offline-meta-reinforcement-learning-with-1","title":"Offline Meta-Reinforcement Learning with Online Self-Supervision","date":"2021-07-08","arxiv_id":"2107.03974","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-online-service-coordination-using","slug":"distributed-online-service-coordination-using","title":"Distributed Online Service Coordination Using Deep Reinforcement Learning","date":"2021-07-07","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/ensemble-and-auxiliary-tasks-for-data","slug":"ensemble-and-auxiliary-tasks-for-data","title":"Ensemble and Auxiliary Tasks for Data-Efficient Deep Reinforcement Learning","date":"2021-07-05","arxiv_id":"2107.01904","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ensemble-and-auxiliary-tasks-for-data#ran","syntology_url":"https://syntology.ai/paper/2107.01904","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.01904"}},"official":{"repos":["NUS-LID/RENAULT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-efficient-reinforcement-learning-via-2","slug":"sample-efficient-reinforcement-learning-via-2","title":"Sample Efficient Reinforcement Learning via Model-Ensemble Exploration and Exploitation","date":"2021-07-05","arxiv_id":"2107.01825","repositories_listed":1,"syntology":null},{"url":"/paper/where-is-the-grass-greener-revisiting","slug":"where-is-the-grass-greener-revisiting","title":"Optimality Inductive Biases and Agnostic Guidelines for Offline Reinforcement Learning","date":"2021-07-03","arxiv_id":"2107.01407","repositories_listed":1,"syntology":null},{"url":"/paper/ensemble-kalman-filter-enkf-for-reinforcement","slug":"ensemble-kalman-filter-enkf-for-reinforcement","title":"Controlled Interacting Particle Algorithms for Simulation-based Reinforcement Learning","date":"2021-07-02","arxiv_id":"2107.01244","repositories_listed":1,"syntology":null},{"url":"/paper/distilling-reinforcement-learning-tricks-for","slug":"distilling-reinforcement-learning-tricks-for","title":"Distilling Reinforcement Learning Tricks for Video Games","date":"2021-07-01","arxiv_id":"2107.00703","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-adversarial-attacks-on-1","slug":"understanding-adversarial-attacks-on-1","title":"Understanding Adversarial Attacks on Observations in Deep Reinforcement Learning","date":"2021-06-30","arxiv_id":"2106.15860","repositories_listed":1,"syntology":null},{"url":"/paper/causal-reinforcement-learning-using","slug":"causal-reinforcement-learning-using","title":"Causal Reinforcement Learning using Observational and Interventional Data","date":"2021-06-28","arxiv_id":"2106.14421","repositories_listed":1,"syntology":null},{"url":"/paper/graph-convolutional-memory-for-deep","slug":"graph-convolutional-memory-for-deep","title":"Graph Convolutional Memory using Topological Priors","date":"2021-06-27","arxiv_id":"2106.14117","repositories_listed":1,"syntology":null},{"url":"/paper/compositional-reinforcement-learning-from","slug":"compositional-reinforcement-learning-from","title":"Compositional Reinforcement Learning from Logical Specifications","date":"2021-06-25","arxiv_id":"2106.13906","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/compositional-reinforcement-learning-from#ran","syntology_url":"https://syntology.ai/paper/2106.13906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.13906"}},"official":{"repos":["keyshor/dirl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-reinforcement-learning-via-latent-1","slug":"model-based-reinforcement-learning-via-latent-1","title":"Model-Based Reinforcement Learning via Latent-Space Collocation","date":"2021-06-24","arxiv_id":"2106.13229","repositories_listed":1,"syntology":null},{"url":"/paper/bregman-gradient-policy-optimization","slug":"bregman-gradient-policy-optimization","title":"Bregman Gradient Policy Optimization","date":"2021-06-23","arxiv_id":"2106.12112","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bregman-gradient-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2106.12112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.12112"}},"official":{"repos":["gaosh/bgpo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-phy-layer","slug":"reinforcement-learning-for-phy-layer","title":"Reinforcement Learning for Physical Layer Communications","date":"2021-06-22","arxiv_id":"2106.11595","repositories_listed":1,"syntology":null},{"url":"/paper/a-max-min-entropy-framework-for-reinforcement","slug":"a-max-min-entropy-framework-for-reinforcement","title":"A Max-Min Entropy Framework for Reinforcement Learning","date":"2021-06-19","arxiv_id":"2106.10517","repositories_listed":1,"syntology":null},{"url":"/paper/towards-safe-reinforcement-learning-via-1","slug":"towards-safe-reinforcement-learning-via-1","title":"Towards Safe Reinforcement Learning via Constraining Conditional Value at Risk","date":"2021-06-18","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-the-weaknesses-of-reinforcement-1","slug":"revisiting-the-weaknesses-of-reinforcement-1","title":"Revisiting the Weaknesses of Reinforcement Learning for Neural Machine Translation","date":"2021-06-16","arxiv_id":"2106.08942","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/revisiting-the-weaknesses-of-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2106.08942","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.08942"}},"official":{"repos":["samuki/reinforce-joey"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-reinforcement-learning-using-advantage","slug":"safe-reinforcement-learning-using-advantage","title":"Safe Reinforcement Learning Using Advantage-Based Intervention","date":"2021-06-16","arxiv_id":"2106.09110","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":6,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safe-reinforcement-learning-using-advantage#ran","syntology_url":"https://syntology.ai/paper/2106.09110","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.09110"}},"official":{"repos":["nolanwagener/safe_rl"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-conservation","slug":"deep-reinforcement-learning-for-conservation","title":"Deep Reinforcement Learning for Conservation Decisions","date":"2021-06-15","arxiv_id":"2106.08272","repositories_listed":1,"syntology":null},{"url":"/paper/randomized-exploration-for-reinforcement","slug":"randomized-exploration-for-reinforcement","title":"Randomized Exploration for Reinforcement Learning with General Value Function Approximation","date":"2021-06-15","arxiv_id":"2106.07841","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/randomized-exploration-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2106.07841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.07841"}},"official":{"repos":["qlan3/Explorer"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-reinforcement-learning-under-minimax","slug":"robust-reinforcement-learning-under-minimax","title":"Robust Reinforcement Learning Under Minimax Regret for Green Security","date":"2021-06-15","arxiv_id":"2106.08413","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-reinforcement-learning-under-minimax#ran","syntology_url":"https://syntology.ai/paper/2106.08413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.08413"}},"official":{"repos":["lily-x/mirror"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rsoccer-a-framework-for-studying","slug":"rsoccer-a-framework-for-studying","title":"rSoccer: A Framework for Studying Reinforcement Learning in Small and Very Small Size Robot Soccer","date":"2021-06-15","arxiv_id":"2106.12895","repositories_listed":1,"syntology":null},{"url":"/paper/learning-intrusion-prevention-policies","slug":"learning-intrusion-prevention-policies","title":"Learning Intrusion Prevention Policies through Optimal Stopping","date":"2021-06-14","arxiv_id":"2106.07160","repositories_listed":1,"syntology":null},{"url":"/paper/contingency-aware-influence-maximization-a","slug":"contingency-aware-influence-maximization-a","title":"Contingency-Aware Influence Maximization: A Reinforcement Learning Approach","date":"2021-06-13","arxiv_id":"2106.07039","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based-group","slug":"deep-reinforcement-learning-based-group","title":"Deep Reinforcement Learning based Group Recommender System","date":"2021-06-13","arxiv_id":"2106.06900","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-as-one-big-sequence-1","slug":"reinforcement-learning-as-one-big-sequence-1","title":"Reinforcement Learning as One Big Sequence Modeling Problem","date":"2021-06-13","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-reinforcement-learning-approach-to-4","slug":"a-deep-reinforcement-learning-approach-to-4","title":"A Deep Reinforcement Learning Approach to Marginalized Importance Sampling with the Successor Representation","date":"2021-06-12","arxiv_id":"2106.06854","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/a-deep-reinforcement-learning-approach-to-4#ran","syntology_url":"https://syntology.ai/paper/2106.06854","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.06854"}},"official":{"repos":["sfujim/SR-DICE"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-game-theoretic-approach-to-multi-agent","slug":"a-game-theoretic-approach-to-multi-agent","title":"A Game-Theoretic Approach to Multi-Agent Trust Region Optimization","date":"2021-06-12","arxiv_id":"2106.06828","repositories_listed":1,"syntology":null},{"url":"/paper/douzero-mastering-doudizhu-with-self-play","slug":"douzero-mastering-doudizhu-with-self-play","title":"DouZero: Mastering DouDizhu with Self-Play Deep Reinforcement Learning","date":"2021-06-11","arxiv_id":"2106.06135","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/douzero-mastering-doudizhu-with-self-play#ran","syntology_url":"https://syntology.ai/paper/2106.06135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.06135"}},"official":{"repos":["kwai/DouZero"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"url":"/paper/wax-ml-a-python-library-for-machine-learning","slug":"wax-ml-a-python-library-for-machine-learning","title":"WAX-ML: A Python library for machine learning and feedback loops on streaming data","date":"2021-06-11","arxiv_id":"2106.06524","repositories_listed":1,"syntology":null},{"url":"/paper/simplifying-deep-reinforcement-learning-via","slug":"simplifying-deep-reinforcement-learning-via","title":"Simplifying Deep Reinforcement Learning via Self-Supervision","date":"2021-06-10","arxiv_id":"2106.05526","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/simplifying-deep-reinforcement-learning-via#ran","syntology_url":"https://syntology.ai/paper/2106.05526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.05526"}},"official":{"repos":["daochenzha/SSRL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/synthesising-reinforcement-learning-policies","slug":"synthesising-reinforcement-learning-policies","title":"Synthesising Reinforcement Learning Policies through Set-Valued Inductive Rule Learning","date":"2021-06-10","arxiv_id":"2106.06009","repositories_listed":1,"syntology":null},{"url":"/paper/pretraining-representations-for-data","slug":"pretraining-representations-for-data","title":"Pretraining Representations for Data-Efficient Reinforcement Learning","date":"2021-06-09","arxiv_id":"2106.04799","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pretraining-representations-for-data#ran","syntology_url":"https://syntology.ai/paper/2106.04799","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04799"}},"official":{"repos":["mila-iqia/SGI"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/self-attention-recurrent-summarization","slug":"self-attention-recurrent-summarization","title":"Self-Attention Recurrent Summarization Network with Reinforcement Learning for Video Summarization Task","date":"2021-06-09","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/self-paced-context-evaluation-for-contextual","slug":"self-paced-context-evaluation-for-contextual","title":"Self-Paced Context Evaluation for Contextual Reinforcement Learning","date":"2021-06-09","arxiv_id":"2106.05110","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-sparse-training-for-deep","slug":"dynamic-sparse-training-for-deep","title":"Dynamic Sparse Training for Deep Reinforcement Learning","date":"2021-06-08","arxiv_id":"2106.04217","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-sparse-training-for-deep#ran","syntology_url":"https://syntology.ai/paper/2106.04217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04217"}},"official":{"repos":["GhadaSokar/Dynamic-Sparse-Training-for-Deep-Reinforcement-Learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-markov-state-abstractions-for-deep","slug":"learning-markov-state-abstractions-for-deep","title":"Learning Markov State Abstractions for Deep Reinforcement Learning","date":"2021-06-08","arxiv_id":"2106.04379","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-markov-state-abstractions-for-deep#ran","syntology_url":"https://syntology.ai/paper/2106.04379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04379"}},"official":{"repos":["camall3n/markov-state-abstractions"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/task-driven-semantic-coding-via-reinforcement","slug":"task-driven-semantic-coding-via-reinforcement","title":"Task-driven Semantic Coding via Reinforcement Learning","date":"2021-06-07","arxiv_id":"2106.03511","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/task-driven-semantic-coding-via-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2106.03511","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03511"}},"official":{"repos":["USTC-IMCL/Task-driven-Semantic-Coding-via-RL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/verifiable-and-compositional-reinforcement","slug":"verifiable-and-compositional-reinforcement","title":"Verifiable and Compositional Reinforcement Learning Systems","date":"2021-06-07","arxiv_id":"2106.05864","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/verifiable-and-compositional-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2106.05864","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.05864"}},"official":{"repos":["cyrusneary/verifiable-compositional-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/xirl-cross-embodiment-inverse-reinforcement","slug":"xirl-cross-embodiment-inverse-reinforcement","title":"XIRL: Cross-embodiment Inverse Reinforcement Learning","date":"2021-06-07","arxiv_id":"2106.03911","repositories_listed":1,"syntology":null},{"url":"/paper/control-oriented-model-based-reinforcement","slug":"control-oriented-model-based-reinforcement","title":"Control-Oriented Model-Based Reinforcement Learning with Implicit Differentiation","date":"2021-06-06","arxiv_id":"2106.03273","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/control-oriented-model-based-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2106.03273","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03273"}},"official":{"repos":["evgenii-nikishin/omd"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/distributional-reinforcement-learning-with-4","slug":"distributional-reinforcement-learning-with-4","title":"Distributional Reinforcement Learning with Unconstrained Monotonic Neural Networks","date":"2021-06-06","arxiv_id":"2106.03228","repositories_listed":1,"syntology":null},{"url":"/paper/same-state-different-task-continual","slug":"same-state-different-task-continual","title":"Same State, Different Task: Continual Reinforcement Learning without Interference","date":"2021-06-05","arxiv_id":"2106.02940","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/same-state-different-task-continual#ran","syntology_url":"https://syntology.ai/paper/2106.02940","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.02940"}},"official":{"repos":["skezle/owl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-agnostic-and-scalable-counterfactual","slug":"model-agnostic-and-scalable-counterfactual","title":"Model-agnostic and Scalable Counterfactual Explanations via Reinforcement Learning","date":"2021-06-04","arxiv_id":"2106.02597","repositories_listed":1,"syntology":null}],"record_sha256":"433ac5546288d7b156cedac2808878e294a4748ee19cb72b351d1e75ce256d03","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}