{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/28","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":28,"pages_in_order":135,"rows_per_page":100,"rows":[2701,2800],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/27","next":"/task/reinforcement-learning-2/papers/29","papers":[{"url":"/paper/guiding-evolutionary-strategies-by","slug":"guiding-evolutionary-strategies-by","title":"Guiding Evolutionary Strategies by Differentiable Robot Simulators","date":"2021-10-01","arxiv_id":"2110.00438","repositories_listed":1,"syntology":null},{"url":"/paper/offline-reinforcement-learning-with-reverse","slug":"offline-reinforcement-learning-with-reverse","title":"Offline Reinforcement Learning with Reverse Model-based Imagination","date":"2021-10-01","arxiv_id":"2110.00188","repositories_listed":1,"syntology":null},{"url":"/paper/molucinate-a-generative-model-for-molecules","slug":"molucinate-a-generative-model-for-molecules","title":"MOLUCINATE: A Generative Model for Molecules in 3D Space","date":"2021-09-30","arxiv_id":"2109.15308","repositories_listed":1,"syntology":null},{"url":"/paper/scalable-online-planning-via-reinforcement","slug":"scalable-online-planning-via-reinforcement","title":"Scalable Online Planning via Reinforcement Learning Fine-Tuning","date":"2021-09-30","arxiv_id":"2109.15316","repositories_listed":1,"syntology":null},{"url":"/paper/surveillance-evasion-through-bayesian","slug":"surveillance-evasion-through-bayesian","title":"Surveillance Evasion Through Bayesian Reinforcement Learning","date":"2021-09-30","arxiv_id":"2109.14811","repositories_listed":1,"syntology":null},{"url":"/paper/unified-data-collection-for-visual-inertial","slug":"unified-data-collection-for-visual-inertial","title":"Unified Data Collection for Visual-Inertial Calibration via Deep Reinforcement Learning","date":"2021-09-30","arxiv_id":"2109.14974","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unified-data-collection-for-visual-inertial#ran","syntology_url":"https://syntology.ai/paper/2109.14974","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.14974"}},"official":{"repos":["ethz-asl/Learn-to-Calibrate"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/width-based-planning-and-active-learning-for","slug":"width-based-planning-and-active-learning-for","title":"Is Policy Learning Overrated?: Width-Based Planning and Active Learning for Atari","date":"2021-09-30","arxiv_id":"2109.15310","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/width-based-planning-and-active-learning-for#ran","syntology_url":"https://syntology.ai/paper/2109.15310","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.15310"}},"official":{"repos":["ibm/atari-active-learning"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fourier-features-in-reinforcement-learning","slug":"fourier-features-in-reinforcement-learning","title":"Fourier Features in Reinforcement Learning with Neural Networks","date":"2021-09-29","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/hyperdqn-a-randomized-exploration-method-for","slug":"hyperdqn-a-randomized-exploration-method-for","title":"HyperDQN: A Randomized Exploration Method for Deep Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/know-your-action-set-learning-action","slug":"know-your-action-set-learning-action","title":"Know Your Action Set: Learning Action Relations for Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/offline-reinforcement-learning-with-in-sample","slug":"offline-reinforcement-learning-with-in-sample","title":"Offline Reinforcement Learning with In-sample Q-Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/minihack-the-planet-a-sandbox-for-open-ended","slug":"minihack-the-planet-a-sandbox-for-open-ended","title":"MiniHack the Planet: A Sandbox for Open-Ended Reinforcement Learning Research","date":"2021-09-27","arxiv_id":"2109.13202","repositories_listed":1,"syntology":null},{"url":"/paper/prioritized-experience-based-reinforcement","slug":"prioritized-experience-based-reinforcement","title":"Prioritized Experience-based Reinforcement Learning with Human Guidance for Autonomous Driving","date":"2021-09-26","arxiv_id":"2109.12516","repositories_listed":1,"syntology":null},{"url":"/paper/emergent-behavior-and-neural-dynamics-in","slug":"emergent-behavior-and-neural-dynamics-in","title":"Emergent behavior and neural dynamics in artificial agents tracking turbulent plumes","date":"2021-09-25","arxiv_id":"2109.12434","repositories_listed":1,"syntology":null},{"url":"/paper/stackelberg-actor-critic-game-theoretic","slug":"stackelberg-actor-critic-game-theoretic","title":"Stackelberg Actor-Critic: Game-Theoretic Reinforcement Learning Algorithms","date":"2021-09-25","arxiv_id":"2109.12286","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stackelberg-actor-critic-game-theoretic#ran","syntology_url":"https://syntology.ai/paper/2109.12286","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.12286"}},"official":{"repos":["leozhengzly/stackelberg-actor-critic-algos"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/nice-robust-scheduling-through-reinforcement","slug":"nice-robust-scheduling-through-reinforcement","title":"NICE: Robust Scheduling through Reinforcement Learning-Guided Integer Programming","date":"2021-09-24","arxiv_id":"2109.12171","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-navigational-safety-in-crowded","slug":"enhancing-navigational-safety-in-crowded","title":"Enhancing Navigational Safety in Crowded Environments using Semantic-Deep-Reinforcement-Learning-based Navigation","date":"2021-09-23","arxiv_id":"2109.11288","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforcement-learning-benchmark-for","slug":"a-reinforcement-learning-benchmark-for","title":"A Reinforcement Learning Benchmark for Autonomous Driving in Intersection Scenarios","date":"2021-09-22","arxiv_id":"2109.10557","repositories_listed":1,"syntology":null},{"url":"/paper/a-workflow-for-offline-model-free-robotic","slug":"a-workflow-for-offline-model-free-robotic","title":"A Workflow for Offline Model-Free Robotic Reinforcement Learning","date":"2021-09-22","arxiv_id":"2109.10813","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-workflow-for-offline-model-free-robotic#ran","syntology_url":"https://syntology.ai/paper/2109.10813","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.10813"}},"official":null}},{"url":"/paper/enero-efficient-real-time-routing","slug":"enero-efficient-real-time-routing","title":"ENERO: Efficient Real-Time WAN Routing Optimization with Deep Reinforcement Learning","date":"2021-09-22","arxiv_id":"2109.10883","repositories_listed":1,"syntology":null},{"url":"/paper/estimation-error-correction-in-deep","slug":"estimation-error-correction-in-deep","title":"Estimation Error Correction in Deep Reinforcement Learning for Deterministic Actor-Critic Methods","date":"2021-09-22","arxiv_id":"2109.10736","repositories_listed":1,"syntology":null},{"url":"/paper/autophoto-aesthetic-photo-capture-using","slug":"autophoto-aesthetic-photo-capture-using","title":"AutoPhoto: Aesthetic Photo Capture using Reinforcement Learning","date":"2021-09-21","arxiv_id":"2109.09923","repositories_listed":1,"syntology":null},{"url":"/paper/deep-policies-for-online-bipartite-matching-a","slug":"deep-policies-for-online-bipartite-matching-a","title":"Deep Policies for Online Bipartite Matching: A Reinforcement Learning Approach","date":"2021-09-21","arxiv_id":"2109.10380","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/deep-policies-for-online-bipartite-matching-a#ran","syntology_url":"https://syntology.ai/paper/2109.10380","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.10380"}},"official":{"repos":["lyeskhalil/corl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/generalization-in-text-based-games-via","slug":"generalization-in-text-based-games-via","title":"Generalization in Text-based Games via Hierarchical Reinforcement Learning","date":"2021-09-21","arxiv_id":"2109.09968","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalization-in-text-based-games-via#ran","syntology_url":"https://syntology.ai/paper/2109.09968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.09968"}},"official":{"repos":["yunqiuxu/h-kga"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hindsight-foresight-relabeling-for-meta","slug":"hindsight-foresight-relabeling-for-meta","title":"Hindsight Foresight Relabeling for Meta-Reinforcement Learning","date":"2021-09-18","arxiv_id":"2109.09031","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hindsight-foresight-relabeling-for-meta#ran","syntology_url":"https://syntology.ai/paper/2109.09031","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.09031"}},"official":{"repos":["michaelwan11/hfr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-state-representation-learning-for","slug":"efficient-state-representation-learning-for","title":"POAR: Efficient Policy Optimization via Online Abstract State Representation Learning","date":"2021-09-17","arxiv_id":"2109.08642","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-robustness-of-distributional","slug":"exploring-the-robustness-of-distributional","title":"Exploring the Training Robustness of Distributional Reinforcement Learning against Noisy State Observations","date":"2021-09-17","arxiv_id":"2109.08776","repositories_listed":1,"syntology":null},{"url":"/paper/back-to-basics-deep-reinforcement-learning-in","slug":"back-to-basics-deep-reinforcement-learning-in","title":"Back to Basics: Deep Reinforcement Learning in Traffic Signal Control","date":"2021-09-15","arxiv_id":"2109.07180","repositories_listed":1,"syntology":null},{"url":"/paper/dcur-data-curriculum-for-teaching-via-samples","slug":"dcur-data-curriculum-for-teaching-via-samples","title":"DCUR: Data Curriculum for Teaching via Samples with Reinforcement Learning","date":"2021-09-15","arxiv_id":"2109.07380","repositories_listed":1,"syntology":null},{"url":"/paper/estimation-of-warfarin-dosage-with","slug":"estimation-of-warfarin-dosage-with","title":"Estimation of Warfarin Dosage with Reinforcement Learning","date":"2021-09-15","arxiv_id":"2109.07564","repositories_listed":1,"syntology":null},{"url":"/paper/automatically-exposing-problems-with-neural","slug":"automatically-exposing-problems-with-neural","title":"Automatically Exposing Problems with Neural Dialog Models","date":"2021-09-14","arxiv_id":"2109.06950","repositories_listed":1,"syntology":null},{"url":"/paper/few-shot-quality-diversity-optimisation","slug":"few-shot-quality-diversity-optimisation","title":"Few-shot Quality-Diversity Optimization","date":"2021-09-14","arxiv_id":"2109.06826","repositories_listed":1,"syntology":null},{"url":"/paper/gradient-imitation-reinforcement-learning-for","slug":"gradient-imitation-reinforcement-learning-for","title":"Gradient Imitation Reinforcement Learning for Low Resource Relation Extraction","date":"2021-09-14","arxiv_id":"2109.06415","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gradient-imitation-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2109.06415","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.06415"}},"official":{"repos":["thu-bpm/gradlre"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-evolutionary","slug":"reinforcement-learning-with-evolutionary","title":"Reinforcement Learning with Evolutionary Trajectory Generator: A General Approach for Quadrupedal Locomotion","date":"2021-09-14","arxiv_id":"2109.06409","repositories_listed":1,"syntology":null},{"url":"/paper/towards-optimized-actions-in-critical","slug":"towards-optimized-actions-in-critical","title":"Towards optimized actions in critical situations of soccer games with deep reinforcement learning","date":"2021-09-14","arxiv_id":"2109.06625","repositories_listed":1,"syntology":null},{"url":"/paper/wavecorr-correlation-savvy-deep-reinforcement","slug":"wavecorr-correlation-savvy-deep-reinforcement","title":"WaveCorr: Correlation-savvy Deep Reinforcement Learning for Portfolio Management","date":"2021-09-14","arxiv_id":"2109.07005","repositories_listed":1,"syntology":null},{"url":"/paper/direct-random-search-for-fine-tuning-of-deep","slug":"direct-random-search-for-fine-tuning-of-deep","title":"Direct Random Search for Fine Tuning of Deep Reinforcement Learning Policies","date":"2021-09-12","arxiv_id":"2109.05604","repositories_listed":1,"syntology":null},{"url":"/paper/hyar-addressing-discrete-continuous-action","slug":"hyar-addressing-discrete-continuous-action","title":"HyAR: Addressing Discrete-Continuous Action Reinforcement Learning via Hybrid Action Representation","date":"2021-09-12","arxiv_id":"2109.05490","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":6,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; every one of the 6 samples that ran constructed an object rather than computing a result","sample_list":"/paper/hyar-addressing-discrete-continuous-action#ran","syntology_url":"https://syntology.ai/paper/2109.05490","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.05490"}},"official":null}},{"url":"/paper/opirl-sample-efficient-off-policy-inverse","slug":"opirl-sample-efficient-off-policy-inverse","title":"OPIRL: Sample Efficient Off-Policy Inverse Reinforcement Learning via Distribution Matching","date":"2021-09-09","arxiv_id":"2109.04307","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/opirl-sample-efficient-off-policy-inverse#ran","syntology_url":"https://syntology.ai/paper/2109.04307","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.04307"}},"official":{"repos":["sff1019/opirl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/timetraveler-reinforcement-learning-for","slug":"timetraveler-reinforcement-learning-for","title":"TimeTraveler: Reinforcement Learning for Temporal Knowledge Graph Forecasting","date":"2021-09-09","arxiv_id":"2109.04101","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/timetraveler-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2109.04101","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.04101"}},"official":{"repos":["jhl-hust/titer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/powergym-a-reinforcement-learning-environment","slug":"powergym-a-reinforcement-learning-environment","title":"PowerGym: A Reinforcement Learning Environment for Volt-Var Control in Power Distribution Systems","date":"2021-09-08","arxiv_id":"2109.03970","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/powergym-a-reinforcement-learning-environment#ran","syntology_url":"https://syntology.ai/paper/2109.03970","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.03970"}},"official":{"repos":["siemens/powergym"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/optimizing-quantum-variational-circuits-with","slug":"optimizing-quantum-variational-circuits-with","title":"Optimizing Quantum Variational Circuits with Deep Reinforcement Learning","date":"2021-09-07","arxiv_id":"2109.03188","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimizing-quantum-variational-circuits-with#ran","syntology_url":"https://syntology.ai/paper/2109.03188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.03188"}},"official":{"repos":["lockwo/rl_qvc_opt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/temporal-aware-deep-reinforcement-learning","slug":"temporal-aware-deep-reinforcement-learning","title":"Temporal Shift Reinforcement Learning","date":"2021-09-05","arxiv_id":"2109.02145","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-multi-latent-space-reinforcement","slug":"unsupervised-multi-latent-space-reinforcement","title":"Unsupervised multi-latent space reinforcement learning framework for video summarization in ultrasound imaging","date":"2021-09-03","arxiv_id":"2109.01309","repositories_listed":1,"syntology":null},{"url":"/paper/a-comparative-study-of-algorithms-for-1","slug":"a-comparative-study-of-algorithms-for-1","title":"A Comparative Study of Algorithms for Intelligent Traffic Signal Control","date":"2021-09-02","arxiv_id":"2109.00937","repositories_listed":1,"syntology":null},{"url":"/paper/catastrophic-interference-in-reinforcement","slug":"catastrophic-interference-in-reinforcement","title":"Catastrophic Interference in Reinforcement Learning: A Solution Based on Context Division and Knowledge Distillation","date":"2021-09-01","arxiv_id":"2109.00525","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/catastrophic-interference-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2109.00525","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.00525"}},"official":{"repos":["sweety-dm/interference-aware-deep-q-learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/identifying-optimal-cycles-in-quantum-thermal","slug":"identifying-optimal-cycles-in-quantum-thermal","title":"Identifying optimal cycles in quantum thermal machines with reinforcement-learning","date":"2021-08-30","arxiv_id":"2108.13525","repositories_listed":1,"syntology":null},{"url":"/paper/regen-reinforcement-learning-for-text-and","slug":"regen-reinforcement-learning-for-text-and","title":"ReGen: Reinforcement Learning for Text and Knowledge Base Generation using Pretrained Language Models","date":"2021-08-27","arxiv_id":"2108.12472","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regen-reinforcement-learning-for-text-and#ran","syntology_url":"https://syntology.ai/paper/2108.12472","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.12472"}},"official":{"repos":["IBM/regen"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-based-condition","slug":"reinforcement-learning-based-condition","title":"Reinforcement Learning based Condition-oriented Maintenance Scheduling for Flow Line Systems","date":"2021-08-27","arxiv_id":"2108.12298","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-powered-semantic","slug":"reinforcement-learning-powered-semantic","title":"Reinforcement Learning-powered Semantic Communication via Semantic Similarity","date":"2021-08-27","arxiv_id":"2108.12121","repositories_listed":1,"syntology":null},{"url":"/paper/responsive-regulation-of-dynamic-uav","slug":"responsive-regulation-of-dynamic-uav","title":"Responsive Regulation of Dynamic UAV Communication Networks Based on Deep Reinforcement Learning","date":"2021-08-25","arxiv_id":"2108.11012","repositories_listed":1,"syntology":null},{"url":"/paper/robust-risk-aware-reinforcement-learning","slug":"robust-risk-aware-reinforcement-learning","title":"Robust Risk-Aware Reinforcement Learning","date":"2021-08-23","arxiv_id":"2108.10403","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-state-augmentation-methods-for","slug":"revisiting-state-augmentation-methods-for","title":"Revisiting State Augmentation methods for Reinforcement Learning with Stochastic Delays","date":"2021-08-17","arxiv_id":"2108.07555","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/revisiting-state-augmentation-methods-for#ran","syntology_url":"https://syntology.ai/paper/2108.07555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.07555"}},"official":{"repos":["baranwa2/delayresolvedrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/aspect-sentiment-triplet-extraction-using","slug":"aspect-sentiment-triplet-extraction-using","title":"Aspect Sentiment Triplet Extraction Using Reinforcement Learning","date":"2021-08-13","arxiv_id":"2108.06107","repositories_listed":1,"syntology":null},{"url":"/paper/q-mixing-network-for-multi-agent-pathfinding","slug":"q-mixing-network-for-multi-agent-pathfinding","title":"Q-Mixing Network for Multi-Agent Pathfinding in Partially Observable Grid Environments","date":"2021-08-13","arxiv_id":"2108.06148","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-robot-navigation","slug":"reinforcement-learning-for-robot-navigation","title":"Reinforcement Learning for Robot Navigation with Adaptive Forward Simulation Time (AFST) in a Semi-Markov Model","date":"2021-08-13","arxiv_id":"2108.06161","repositories_listed":1,"syntology":null},{"url":"/paper/a-functional-mirror-ascent-view-of-policy","slug":"a-functional-mirror-ascent-view-of-policy","title":"A general class of surrogate functions for stable and efficient reinforcement learning","date":"2021-08-12","arxiv_id":"2108.05828","repositories_listed":1,"syntology":null},{"url":"/paper/fairness-through-counterfactual-utilities","slug":"fairness-through-counterfactual-utilities","title":"Fairness Through Counterfactual Utilities","date":"2021-08-11","arxiv_id":"2108.05315","repositories_listed":1,"syntology":null},{"url":"/paper/gap-dependent-unsupervised-exploration-for","slug":"gap-dependent-unsupervised-exploration-for","title":"Gap-Dependent Unsupervised Exploration for Reinforcement Learning","date":"2021-08-11","arxiv_id":"2108.05439","repositories_listed":1,"syntology":null},{"url":"/paper/imitation-learning-by-reinforcement-learning","slug":"imitation-learning-by-reinforcement-learning","title":"Imitation Learning by Reinforcement Learning","date":"2021-08-10","arxiv_id":"2108.04763","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imitation-learning-by-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2108.04763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.04763"}},"official":{"repos":["spotify-research/il-by-rl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-deep-reinforcement-learning-for-multi","slug":"safe-deep-reinforcement-learning-for-multi","title":"Safe Deep Reinforcement Learning for Multi-Agent Systems with Continuous Action Spaces","date":"2021-08-09","arxiv_id":"2108.03952","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safe-deep-reinforcement-learning-for-multi#ran","syntology_url":"https://syntology.ai/paper/2108.03952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.03952"}},"official":{"repos":["zisikons/deep-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/verlpy-python-library-for-verification-of","slug":"verlpy-python-library-for-verification-of","title":"VeRLPy: Python Library for Verification of Digital Designs with Reinforcement Learning","date":"2021-08-09","arxiv_id":"2108.03978","repositories_listed":1,"syntology":null},{"url":"/paper/meta-reinforcement-learning-in-broad-and-non","slug":"meta-reinforcement-learning-in-broad-and-non","title":"Meta-Reinforcement Learning in Broad and Non-Parametric Environments","date":"2021-08-08","arxiv_id":"2108.03718","repositories_listed":1,"syntology":null},{"url":"/paper/what-matters-in-learning-from-offline-human","slug":"what-matters-in-learning-from-offline-human","title":"What Matters in Learning from Offline Human Demonstrations for Robot Manipulation","date":"2021-08-06","arxiv_id":"2108.03298","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/what-matters-in-learning-from-offline-human#ran","syntology_url":"https://syntology.ai/paper/2108.03298","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.03298"}},"official":null}},{"url":"/paper/an-encoder-decoder-based-audio-captioning","slug":"an-encoder-decoder-based-audio-captioning","title":"An Encoder-Decoder Based Audio Captioning System With Transfer and Reinforcement Learning","date":"2021-08-05","arxiv_id":"2108.02752","repositories_listed":1,"syntology":null},{"url":"/paper/the-ai-economist-optimal-economic-policy","slug":"the-ai-economist-optimal-economic-policy","title":"The AI Economist: Optimal Economic Policy Design via Two-level Deep Reinforcement Learning","date":"2021-08-05","arxiv_id":"2108.02755","repositories_listed":1,"syntology":null},{"url":"/paper/a-pragmatic-look-at-deep-imitation-learning","slug":"a-pragmatic-look-at-deep-imitation-learning","title":"A Pragmatic Look at Deep Imitation Learning","date":"2021-08-04","arxiv_id":"2108.01867","repositories_listed":1,"syntology":null},{"url":"/paper/learning-barrier-certificates-towards-safe","slug":"learning-barrier-certificates-towards-safe","title":"Learning Barrier Certificates: Towards Safe Reinforcement Learning with Zero Training-time Violations","date":"2021-08-04","arxiv_id":"2108.01846","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-barrier-certificates-towards-safe#ran","syntology_url":"https://syntology.ai/paper/2108.01846","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.01846"}},"official":{"repos":["roosephu/crabs"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/learning-task-agnostic-skills-with-data","slug":"learning-task-agnostic-skills-with-data","title":"Learning Task Agnostic Skills with Data-driven Guidance","date":"2021-08-04","arxiv_id":"2108.01869","repositories_listed":1,"syntology":null},{"url":"/paper/a-scalable-federated-multi-agent-architecture","slug":"a-scalable-federated-multi-agent-architecture","title":"Scalable Multi-agent Reinforcement Learning Algorithm for Wireless Networks","date":"2021-08-01","arxiv_id":"2108.00506","repositories_listed":1,"syntology":null},{"url":"/paper/how-helpful-is-inverse-reinforcement-learning","slug":"how-helpful-is-inverse-reinforcement-learning","title":"How Helpful is Inverse Reinforcement Learning for Table-to-Text Generation?","date":"2021-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/phrase-level-action-reinforcement-learning","slug":"phrase-level-action-reinforcement-learning","title":"Phrase-Level Action Reinforcement Learning for Neural Dialog Response Generation","date":"2021-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/adaptable-image-quality-assessment-using-meta","slug":"adaptable-image-quality-assessment-using-meta","title":"Adaptable image quality assessment using meta-reinforcement learning of task amenability","date":"2021-07-31","arxiv_id":"2108.04359","repositories_listed":1,"syntology":null},{"url":"/paper/rltutor-reinforcement-learning-based-adaptive","slug":"rltutor-reinforcement-learning-based-adaptive","title":"RLTutor: Reinforcement Learning Based Adaptive Tutoring System by Modeling Virtual Student with Fewer Interactions","date":"2021-07-31","arxiv_id":"2108.00268","repositories_listed":1,"syntology":null},{"url":"/paper/sequence-adaptation-via-reinforcement","slug":"sequence-adaptation-via-reinforcement","title":"Sequence Adaptation via Reinforcement Learning in Recommender Systems","date":"2021-07-31","arxiv_id":"2108.01442","repositories_listed":1,"syntology":null},{"url":"/paper/strategically-efficient-exploration-in","slug":"strategically-efficient-exploration-in","title":"Strategically Efficient Exploration in Competitive Multi-agent Reinforcement Learning","date":"2021-07-30","arxiv_id":"2107.14698","repositories_listed":1,"syntology":null},{"url":"/paper/controlling-epidemics-through-optimal","slug":"controlling-epidemics-through-optimal","title":"Controlling epidemics through optimal allocation of test kits and vaccine doses across networks","date":"2021-07-29","arxiv_id":"2107.13709","repositories_listed":1,"syntology":null},{"url":"/paper/tianshou-a-highly-modularized-deep","slug":"tianshou-a-highly-modularized-deep","title":"Tianshou: a Highly Modularized Deep Reinforcement Learning Library","date":"2021-07-29","arxiv_id":"2107.14171","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tianshou-a-highly-modularized-deep#ran","syntology_url":"https://syntology.ai/paper/2107.14171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.14171"}},"official":{"repos":["thu-ml/tianshou"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/finding-failures-in-high-fidelity-simulation","slug":"finding-failures-in-high-fidelity-simulation","title":"Finding Failures in High-Fidelity Simulation using Adaptive Stress Testing and the Backward Algorithm","date":"2021-07-27","arxiv_id":"2107.12940","repositories_listed":1,"syntology":null},{"url":"/paper/model-selection-for-offline-reinforcement","slug":"model-selection-for-offline-reinforcement","title":"Model Selection for Offline Reinforcement Learning: Practical Considerations for Healthcare Settings","date":"2021-07-23","arxiv_id":"2107.11003","repositories_listed":1,"syntology":null},{"url":"/paper/accelerating-quadratic-optimization-with","slug":"accelerating-quadratic-optimization-with","title":"Accelerating Quadratic Optimization with Reinforcement Learning","date":"2021-07-22","arxiv_id":"2107.10847","repositories_listed":1,"syntology":null},{"url":"/paper/demonstration-guided-reinforcement-learning","slug":"demonstration-guided-reinforcement-learning","title":"Demonstration-Guided Reinforcement Learning with Learned Skills","date":"2021-07-21","arxiv_id":"2107.10253","repositories_listed":1,"syntology":null},{"url":"/paper/toward-collaborative-reinforcement-learning","slug":"toward-collaborative-reinforcement-learning","title":"Toward Collaborative Reinforcement Learning Agents that Communicate Through Text-Based Natural Language","date":"2021-07-20","arxiv_id":"2107.09356","repositories_listed":1,"syntology":null},{"url":"/paper/a-sustainable-ecosystem-through-emergent","slug":"a-sustainable-ecosystem-through-emergent","title":"A Sustainable Ecosystem through Emergent Cooperation in Multi-Agent Reinforcement Learning","date":"2021-07-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/decoupling-exploration-and-exploitation-in","slug":"decoupling-exploration-and-exploitation-in","title":"Decoupled Reinforcement Learning to Stabilise Intrinsically-Motivated Exploration","date":"2021-07-19","arxiv_id":"2107.08966","repositories_listed":1,"syntology":null},{"url":"/paper/vision-based-autonomous-car-racing-using-deep","slug":"vision-based-autonomous-car-racing-using-deep","title":"Vision-Based Autonomous Car Racing Using Deep Imitative Reinforcement Learning","date":"2021-07-18","arxiv_id":"2107.08325","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-reinforcement-learning-with-2","slug":"hierarchical-reinforcement-learning-with-2","title":"Hierarchical Reinforcement Learning with Optimal Level Synchronization based on a Deep Generative Model","date":"2021-07-17","arxiv_id":"2107.08183","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforcement-learning-environment-for-2","slug":"a-reinforcement-learning-environment-for-2","title":"A Reinforcement Learning Environment for Mathematical Reasoning via Program Synthesis","date":"2021-07-15","arxiv_id":"2107.07373","repositories_listed":1,"syntology":null},{"url":"/paper/pc-mlp-model-based-reinforcement-learning","slug":"pc-mlp-model-based-reinforcement-learning","title":"PC-MLP: Model-based Reinforcement Learning with Policy Cover Guided Exploration","date":"2021-07-15","arxiv_id":"2107.07410","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pc-mlp-model-based-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2107.07410","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.07410"}},"official":{"repos":["yudasong/PCMLP"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-adaptive-multi-intention-inverse","slug":"deep-adaptive-multi-intention-inverse","title":"Deep Adaptive Multi-Intention Inverse Reinforcement Learning","date":"2021-07-14","arxiv_id":"2107.06692","repositories_listed":1,"syntology":null},{"url":"/paper/safer-reinforcement-learning-through","slug":"safer-reinforcement-learning-through","title":"Safer Reinforcement Learning through Transferable Instinct Networks","date":"2021-07-14","arxiv_id":"2107.06686","repositories_listed":1,"syntology":null},{"url":"/paper/carle-s-game-an-open-ended-challenge-in","slug":"carle-s-game-an-open-ended-challenge-in","title":"Carle's Game: An Open-Ended Challenge in Exploratory Machine Creativity","date":"2021-07-13","arxiv_id":"2107.05786","repositories_listed":1,"syntology":null},{"url":"/paper/rellie-deep-reinforcement-learning-for","slug":"rellie-deep-reinforcement-learning-for","title":"ReLLIE: Deep Reinforcement Learning for Customized Low-Light Image Enhancement","date":"2021-07-13","arxiv_id":"2107.05830","repositories_listed":1,"syntology":null},{"url":"/paper/shortest-path-constrained-reinforcement-1","slug":"shortest-path-constrained-reinforcement-1","title":"Shortest-Path Constrained Reinforcement Learning for Sparse Reward Tasks","date":"2021-07-13","arxiv_id":"2107.06405","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shortest-path-constrained-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2107.06405","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.06405"}},"official":{"repos":["srsohn/shortest-path-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/behavior-constraining-in-weight-space-for","slug":"behavior-constraining-in-weight-space-for","title":"Behavior Constraining in Weight Space for Offline Reinforcement Learning","date":"2021-07-12","arxiv_id":"2107.05479","repositories_listed":1,"syntology":null},{"url":"/paper/conservative-offline-distributional","slug":"conservative-offline-distributional","title":"Conservative Offline Distributional Reinforcement Learning","date":"2021-07-12","arxiv_id":"2107.06106","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/conservative-offline-distributional#ran","syntology_url":"https://syntology.ai/paper/2107.06106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.06106"}},"official":{"repos":["JasonMa2016/CODAC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/explore-and-control-with-adversarial-surprise","slug":"explore-and-control-with-adversarial-surprise","title":"Explore and Control with Adversarial Surprise","date":"2021-07-12","arxiv_id":"2107.07394","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/explore-and-control-with-adversarial-surprise#ran","syntology_url":"https://syntology.ai/paper/2107.07394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.07394"}},"official":null}},{"url":"/paper/modeling-explicit-concerning-states-for","slug":"modeling-explicit-concerning-states-for","title":"Modeling Explicit Concerning States for Reinforcement Learning in Visual Dialogue","date":"2021-07-12","arxiv_id":"2107.05250","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/modeling-explicit-concerning-states-for#ran","syntology_url":"https://syntology.ai/paper/2107.05250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.05250"}},"official":{"repos":["zipengxuc/ecs-visdial-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-better-laplacian-representation-in","slug":"towards-better-laplacian-representation-in","title":"Towards Better Laplacian Representation in Reinforcement Learning with Generalized Graph Drawing","date":"2021-07-12","arxiv_id":"2107.05545","repositories_listed":1,"syntology":null},{"url":"/paper/backprop-free-reinforcement-learning-with","slug":"backprop-free-reinforcement-learning-with","title":"Backprop-Free Reinforcement Learning with Active Neural Generative Coding","date":"2021-07-10","arxiv_id":"2107.07046","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/backprop-free-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2107.07046","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.07046"}},"official":{"repos":["ago109/active-neural-generative-coding"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}}],"record_sha256":"ea473c6fe345d15bb6ae114b15c1f0a7a125efe6c964afc77f1b24b17e6b172e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}