{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/atari-games/papers/3","list_of":"/task/atari-games","task":"Atari Games","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":7,"rows_per_page":100,"rows":[201,300],"of":625,"counts":{"archive_papers_tagged":625,"with_a_code_link":313,"where_syntology_ran_a_sample":118,"not_listed_spam_title":0,"listed":625,"listed_where_code_ran":118,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":97,"every_run_a_failure_of_syntologys_instrument":21,"listed_with_a_run_with_no_instrument_failure":97,"listed_every_run_a_failure_of_syntologys_instrument":21,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/atari-games","prev":"/task/atari-games/papers/2","next":"/task/atari-games/papers/4","papers":[{"url":"/paper/pretraining-representations-for-data","slug":"pretraining-representations-for-data","title":"Pretraining Representations for Data-Efficient Reinforcement Learning","date":"2021-06-09","arxiv_id":"2106.04799","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pretraining-representations-for-data#ran","syntology_url":"https://syntology.ai/paper/2106.04799","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04799"}},"official":{"repos":["mila-iqia/SGI"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/malib-a-parallel-framework-for-population","slug":"malib-a-parallel-framework-for-population","title":"MALib: A Parallel Framework for Population-based Multi-agent Reinforcement Learning","date":"2021-06-05","arxiv_id":"2106.07551","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/malib-a-parallel-framework-for-population#ran","syntology_url":"https://syntology.ai/paper/2106.07551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.07551"}},"official":{"repos":["sjtu-marl/malib"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/provable-representation-learning-for","slug":"provable-representation-learning-for","title":"Provable Representation Learning for Imitation with Contrastive Fourier Features","date":"2021-05-26","arxiv_id":"2105.12272","repositories_listed":1,"syntology":null},{"url":"/paper/adapting-to-reward-progressivity-via-spectral-1","slug":"adapting-to-reward-progressivity-via-spectral-1","title":"Adapting to Reward Progressivity via Spectral Reinforcement Learning","date":"2021-04-29","arxiv_id":"2104.14138","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/adapting-to-reward-progressivity-via-spectral-1#ran","syntology_url":"https://syntology.ai/paper/2104.14138","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.14138"}},"official":{"repos":["mchldann/SpectralDQN"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-on-a-budget-via-teacher-imitation","slug":"learning-on-a-budget-via-teacher-imitation","title":"Learning on a Budget via Teacher Imitation","date":"2021-04-17","arxiv_id":"2104.08440","repositories_listed":1,"syntology":null},{"url":"/paper/a-coevolutionairy-approach-to-deep-multi","slug":"a-coevolutionairy-approach-to-deep-multi","title":"A coevolutionary approach to deep multi-agent reinforcement learning","date":"2021-04-12","arxiv_id":"2104.05610","repositories_listed":1,"syntology":null},{"url":"/paper/generalizable-episodic-memory-for-deep","slug":"generalizable-episodic-memory-for-deep","title":"Generalizable Episodic Memory for Deep Reinforcement Learning","date":"2021-03-11","arxiv_id":"2103.06469","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalizable-episodic-memory-for-deep#ran","syntology_url":"https://syntology.ai/paper/2103.06469","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.06469"}},"official":{"repos":["MouseHu/GEM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/behavior-from-the-void-unsupervised-active","slug":"behavior-from-the-void-unsupervised-active","title":"Behavior From the Void: Unsupervised Active Pre-Training","date":"2021-03-08","arxiv_id":"2103.04551","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/behavior-from-the-void-unsupervised-active#ran","syntology_url":"https://syntology.ai/paper/2103.04551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.04551"}},"official":{"repos":["rll-research/url_benchmark"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/conservative-optimistic-policy-optimization","slug":"conservative-optimistic-policy-optimization","title":"Conservative Optimistic Policy Optimization via Multiple Importance Sampling","date":"2021-03-04","arxiv_id":"2103.03307","repositories_listed":1,"syntology":null},{"url":"/paper/improving-computational-efficiency-in-visual","slug":"improving-computational-efficiency-in-visual","title":"Improving Computational Efficiency in Visual Reinforcement Learning via Stored Embeddings","date":"2021-03-04","arxiv_id":"2103.02886","repositories_listed":1,"syntology":null},{"url":"/paper/q-value-weighted-regression-reinforcement-1","slug":"q-value-weighted-regression-reinforcement-1","title":"Q-Value Weighted Regression: Reinforcement Learning with Limited Data","date":"2021-02-12","arxiv_id":"2102.06782","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/q-value-weighted-regression-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2102.06782","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.06782"}},"official":null}},{"url":"/paper/learning-state-representations-from-random","slug":"learning-state-representations-from-random","title":"Learning State Representations from Random Deep Action-conditional Predictions","date":"2021-02-09","arxiv_id":"2102.04897","repositories_listed":1,"syntology":null},{"url":"/paper/shielding-atari-games-with-bounded-prescience","slug":"shielding-atari-games-with-bounded-prescience","title":"Shielding Atari Games with Bounded Prescience","date":"2021-01-20","arxiv_id":"2101.08153","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-perturbation-based-saliency-maps","slug":"benchmarking-perturbation-based-saliency-maps","title":"Benchmarking Perturbation-based Saliency Maps for Explaining Atari Agents","date":"2021-01-18","arxiv_id":"2101.07312","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-width-based-planning-and","slug":"hierarchical-width-based-planning-and","title":"Hierarchical Width-Based Planning and Learning","date":"2021-01-15","arxiv_id":"2101.06177","repositories_listed":1,"syntology":null},{"url":"/paper/developing-an-openai-gym-compatible-framework","slug":"developing-an-openai-gym-compatible-framework","title":"Developing an OpenAI Gym-compatible framework and simulation environment for testing Deep Reinforcement Learning agents solving the Ambulance Location Problem","date":"2021-01-12","arxiv_id":"2101.04434","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-trust-region-learning","slug":"multi-agent-trust-region-learning","title":"Multi-Agent Trust Region Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-task-clustering-for-multi-task","slug":"unsupervised-task-clustering-for-multi-task","title":"Unsupervised Task Clustering for Multi-Task Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-agents-without-rewards-1","slug":"evaluating-agents-without-rewards-1","title":"Evaluating Agents without Rewards","date":"2020-12-21","arxiv_id":"2012.11538","repositories_listed":1,"syntology":null},{"url":"/paper/high-throughput-synchronous-deep-rl-1","slug":"high-throughput-synchronous-deep-rl-1","title":"High-Throughput Synchronous Deep RL","date":"2020-12-17","arxiv_id":"2012.09849","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/high-throughput-synchronous-deep-rl-1#ran","syntology_url":"https://syntology.ai/paper/2012.09849","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.09849"}},"official":{"repos":["IouJenLiu/HTS-RL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/optimizing-the-neural-architecture-of","slug":"optimizing-the-neural-architecture-of","title":"Optimizing the Neural Architecture of Reinforcement Learning Agents","date":"2020-11-30","arxiv_id":"2011.14632","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-represent-action-values-as-a-1","slug":"learning-to-represent-action-values-as-a-1","title":"Learning to Represent Action Values as a Hypergraph on the Action Vertices","date":"2020-10-28","arxiv_id":"2010.14680","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-represent-action-values-as-a-1#ran","syntology_url":"https://syntology.ai/paper/2010.14680","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.14680"}},"official":{"repos":["atavakol/action-hypergraph-networks"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/masked-contrastive-representation-learning","slug":"masked-contrastive-representation-learning","title":"Masked Contrastive Representation Learning for Reinforcement Learning","date":"2020-10-15","arxiv_id":"2010.07470","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/masked-contrastive-representation-learning#ran","syntology_url":"https://syntology.ai/paper/2010.07470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.07470"}},"official":{"repos":["teslacool/m-curl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/student-initiated-action-advising-via-advice","slug":"student-initiated-action-advising-via-advice","title":"Student-Initiated Action Advising via Advice Novelty","date":"2020-10-01","arxiv_id":"2010.00381","repositories_listed":1,"syntology":null},{"url":"/paper/lucid-dreaming-for-experience-replay","slug":"lucid-dreaming-for-experience-replay","title":"Lucid Dreaming for Experience Replay: Refreshing Past States with the Current Policy","date":"2020-09-29","arxiv_id":"2009.13736","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-variational-model-based","slug":"contrastive-variational-model-based","title":"Contrastive Variational Reinforcement Learning for Complex Observations","date":"2020-08-06","arxiv_id":"2008.02430","repositories_listed":1,"syntology":null},{"url":"/paper/distributional-reinforcement-learning-with-3","slug":"distributional-reinforcement-learning-with-3","title":"Distributional Reinforcement Learning via Moment Matching","date":"2020-07-24","arxiv_id":"2007.12354","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/distributional-reinforcement-learning-with-3#ran","syntology_url":"https://syntology.ai/paper/2007.12354","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.12354"}},"official":{"repos":["thanhnguyentang/mmdrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dinerdash-gym-a-benchmark-for-policy-learning","slug":"dinerdash-gym-a-benchmark-for-policy-learning","title":"DinerDash Gym: A Benchmark for Policy Learning in High-Dimensional Action Space","date":"2020-07-13","arxiv_id":"2007.06207","repositories_listed":1,"syntology":null},{"url":"/paper/learning-abstract-models-for-strategic","slug":"learning-abstract-models-for-strategic","title":"Learning Abstract Models for Strategic Exploration and Fast Reward Transfer","date":"2020-07-12","arxiv_id":"2007.05896","repositories_listed":1,"syntology":null},{"url":"/paper/explanation-augmented-feedback-in-human-in","slug":"explanation-augmented-feedback-in-human-in","title":"Widening the Pipeline in Human-Guided Reinforcement Learning with Explanation and Context-Aware Data Augmentation","date":"2020-06-26","arxiv_id":"2006.14804","repositories_listed":1,"syntology":null},{"url":"/paper/intrinsic-reward-driven-imitation-learning","slug":"intrinsic-reward-driven-imitation-learning","title":"Intrinsic Reward Driven Imitation Learning via Generative Model","date":"2020-06-26","arxiv_id":"2006.15061","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-finite-state-representations-of","slug":"understanding-finite-state-representations-of","title":"Re-understanding Finite-State Representations of Recurrent Policy Networks","date":"2020-06-06","arxiv_id":"2006.03745","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/understanding-finite-state-representations-of#ran","syntology_url":"https://syntology.ai/paper/2006.03745","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.03745"}},"official":{"repos":["modanesh/Differential_IG"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/anomaly-detection-in-video-games","slug":"anomaly-detection-in-video-games","title":"A Metric Learning Approach to Anomaly Detection in Video Games","date":"2020-05-20","arxiv_id":"2005.10211","repositories_listed":1,"syntology":null},{"url":"/paper/local-and-global-explanations-of-agent","slug":"local-and-global-explanations-of-agent","title":"Local and Global Explanations of Agent Behavior: Integrating Strategy Summaries with Saliency Maps","date":"2020-05-18","arxiv_id":"2005.08874","repositories_listed":1,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/local-and-global-explanations-of-agent#ran","syntology_url":"https://syntology.ai/paper/2005.08874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.08874"}},"official":{"repos":["HuTobias/HIGHLIGHTS-LRP"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/explain-your-move-understanding-agent-actions","slug":"explain-your-move-understanding-agent-actions","title":"Explain Your Move: Understanding Agent Actions Using Focused Feature Saliency","date":"2020-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/self-punishment-and-reward-backfill-for-deep","slug":"self-punishment-and-reward-backfill-for-deep","title":"Self Punishment and Reward Backfill for Deep Q-Learning","date":"2020-04-10","arxiv_id":"2004.05002","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-discovering-of-causal","slug":"self-supervised-discovering-of-causal","title":"Self-Supervised Discovering of Interpretable Features for Reinforcement Learning","date":"2020-03-16","arxiv_id":"2003.07069","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-supervised-discovering-of-causal#ran","syntology_url":"https://syntology.ai/paper/2003.07069","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.07069"}},"official":{"repos":["shiwj16/SSINet"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-catastrophic-interference-in-atari-2600","slug":"on-catastrophic-interference-in-atari-2600","title":"On Catastrophic Interference in Atari 2600 Games","date":"2020-02-28","arxiv_id":"2002.12499","repositories_listed":1,"syntology":null},{"url":"/paper/conqur-mitigating-delusional-bias-in-deep-q-1","slug":"conqur-mitigating-delusional-bias-in-deep-q-1","title":"ConQUR: Mitigating Delusional Bias in Deep Q-learning","date":"2020-02-27","arxiv_id":"2002.12399","repositories_listed":1,"syntology":null},{"url":"/paper/discriminative-particle-filter-reinforcement-1","slug":"discriminative-particle-filter-reinforcement-1","title":"Discriminative Particle Filter Reinforcement Learning for Complex Partial Observations","date":"2020-02-23","arxiv_id":"2002.09884","repositories_listed":1,"syntology":null},{"url":"/paper/safe-imitation-learning-via-fast-bayesian","slug":"safe-imitation-learning-via-fast-bayesian","title":"Safe Imitation Learning via Fast Bayesian Reward Inference from Preferences","date":"2020-02-21","arxiv_id":"2002.09089","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/safe-imitation-learning-via-fast-bayesian#ran","syntology_url":"https://syntology.ai/paper/2002.09089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.09089"}},"official":{"repos":["dsbrown1331/bayesianrex"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["community","official"]}}},{"url":"/paper/an-optimistic-perspective-on-offline-deep","slug":"an-optimistic-perspective-on-offline-deep","title":"An Optimistic Perspective on Offline Deep Reinforcement Learning","date":"2020-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/slm-lab-a-comprehensive-benchmark-and-modular-1","slug":"slm-lab-a-comprehensive-benchmark-and-modular-1","title":"SLM Lab: A Comprehensive Benchmark and Modular Software Framework for Reproducible Deep Reinforcement Learning","date":"2019-12-28","arxiv_id":"1912.12482","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/slm-lab-a-comprehensive-benchmark-and-modular-1#ran","syntology_url":"https://syntology.ai/paper/1912.12482","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.12482"}},"official":{"repos":["kengz/SLM-Lab"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/exploratory-not-explanatory-counterfactual-1","slug":"exploratory-not-explanatory-counterfactual-1","title":"Exploratory Not Explanatory: Counterfactual Analysis of Saliency Maps for Deep Reinforcement Learning","date":"2019-12-09","arxiv_id":"1912.05743","repositories_listed":1,"syntology":null},{"url":"/paper/propagating-uncertainty-in-reinforcement","slug":"propagating-uncertainty-in-reinforcement","title":"Propagating Uncertainty in Reinforcement Learning via Wasserstein Barycenters","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reconciling-returns-with-experience-replay","slug":"reconciling-returns-with-experience-replay","title":"Reconciling λ-Returns with Experience Replay","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/memory-efficient-episodic-control","slug":"memory-efficient-episodic-control","title":"Memory-Efficient Episodic Control Reinforcement Learning with Dynamic Online k-means","date":"2019-11-21","arxiv_id":"1911.09560","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-reinforcement-learning-with-2","slug":"sample-efficient-reinforcement-learning-with-2","title":"Sample-Efficient Reinforcement Learning with Maximum Entropy Mellowmax Episodic Control","date":"2019-11-21","arxiv_id":"1911.09615","repositories_listed":1,"syntology":null},{"url":"/paper/momentum-based-accelerated-q-learning","slug":"momentum-based-accelerated-q-learning","title":"Momentum-based Accelerated Q-learning","date":"2019-10-23","arxiv_id":"1910.11673","repositories_listed":1,"syntology":null},{"url":"/paper/formal-language-constraints-for-markov","slug":"formal-language-constraints-for-markov","title":"Formal Language Constraints for Markov Decision Processes","date":"2019-10-02","arxiv_id":"1910.01074","repositories_listed":1,"syntology":null},{"url":"/paper/harnessing-structures-for-value-based","slug":"harnessing-structures-for-value-based","title":"Harnessing Structures for Value-Based Planning and Reinforcement Learning","date":"2019-09-26","arxiv_id":"1909.12255","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/harnessing-structures-for-value-based#ran","syntology_url":"https://syntology.ai/paper/1909.12255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.12255"}},"official":{"repos":["YyzHarry/SV-RL"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/demystifying-active-inference","slug":"demystifying-active-inference","title":"Active inference: demystified and compared","date":"2019-09-24","arxiv_id":"1909.10863","repositories_listed":1,"syntology":null},{"url":"/paper/reusing-convolutional-activations-from-frame","slug":"reusing-convolutional-activations-from-frame","title":"Reusing Convolutional Activations from Frame to Frame to Speed up Training and Inference","date":"2019-09-02","arxiv_id":"1909.05632","repositories_listed":1,"syntology":null},{"url":"/paper/is-deep-reinforcement-learning-really","slug":"is-deep-reinforcement-learning-really","title":"Is Deep Reinforcement Learning Really Superhuman on Atari? Leveling the playing field","date":"2019-08-13","arxiv_id":"1908.04683","repositories_listed":1,"syntology":null},{"url":"/paper/hindsight-trust-region-policy-optimization","slug":"hindsight-trust-region-policy-optimization","title":"Hindsight Trust Region Policy Optimization","date":"2019-07-29","arxiv_id":"1907.12439","repositories_listed":1,"syntology":null},{"url":"/paper/characterizing-attacks-on-deep-reinforcement","slug":"characterizing-attacks-on-deep-reinforcement","title":"Characterizing Attacks on Deep Reinforcement Learning","date":"2019-07-21","arxiv_id":"1907.09470","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/characterizing-attacks-on-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1907.09470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.09470"}},"official":null}},{"url":"/paper/striving-for-simplicity-in-off-policy-deep","slug":"striving-for-simplicity-in-off-policy-deep","title":"An Optimistic Perspective on Offline Reinforcement Learning","date":"2019-07-10","arxiv_id":"1907.04543","repositories_listed":1,"syntology":null},{"url":"/paper/learning-powerful-policies-by-using","slug":"learning-powerful-policies-by-using","title":"Learning Powerful Policies by Using Consistent Dynamics Model","date":"2019-06-11","arxiv_id":"1906.04355","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-low-complexity","slug":"reinforcement-learning-with-low-complexity","title":"Reinforcement Learning with Low-Complexity Liquid State Machines","date":"2019-06-04","arxiv_id":"1906.01695","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reinforcement-learning-with-low-complexity#ran","syntology_url":"https://syntology.ai/paper/1906.01695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.01695"}},"official":{"repos":["wponghiran/lsm-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/exploration-via-flow-based-intrinsic-rewards","slug":"exploration-via-flow-based-intrinsic-rewards","title":"Exploration via Flow-Based Intrinsic Rewards","date":"2019-05-24","arxiv_id":"1905.10071","repositories_listed":1,"syntology":null},{"url":"/paper/combining-experience-replay-with-exploration","slug":"combining-experience-replay-with-exploration","title":"Combining Experience Replay with Exploration by Random Network Distillation","date":"2019-05-18","arxiv_id":"1905.07579","repositories_listed":1,"syntology":null},{"url":"/paper/deep-policies-for-width-based-planning-in","slug":"deep-policies-for-width-based-planning-in","title":"Deep Policies for Width-Based Planning in Pixel Domains","date":"2019-04-12","arxiv_id":"1904.07091","repositories_listed":1,"syntology":null},{"url":"/paper/lets-play-again-variability-of-deep","slug":"lets-play-again-variability-of-deep","title":"Let's Play Again: Variability of Deep Reinforcement Learning Agents in Atari Environments","date":"2019-04-12","arxiv_id":"1904.06312","repositories_listed":1,"syntology":null},{"url":"/paper/jointly-pre-training-with-supervised","slug":"jointly-pre-training-with-supervised","title":"Jointly Pre-training with Supervised, Autoencoder, and Value Losses for Deep Reinforcement Learning","date":"2019-04-03","arxiv_id":"1904.02206","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-dynamic-boltzmann","slug":"reinforcement-learning-with-dynamic-boltzmann","title":"Reinforcement Learning with Dynamic Boltzmann Softmax Updates","date":"2019-03-14","arxiv_id":"1903.05926","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-reinforcement-learning-with-expert","slug":"hybrid-reinforcement-learning-with-expert","title":"Hybrid Reinforcement Learning with Expert State Sequences","date":"2019-03-11","arxiv_id":"1903.04110","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hybrid-reinforcement-learning-with-expert#ran","syntology_url":"https://syntology.ai/paper/1903.04110","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.04110"}},"official":{"repos":["XiaoxiaoGuo/tensor4rl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/using-natural-language-for-reward-shaping-in","slug":"using-natural-language-for-reward-shaping-in","title":"Using Natural Language for Reward Shaping in Reinforcement Learning","date":"2019-03-05","arxiv_id":"1903.02020","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/using-natural-language-for-reward-shaping-in#ran","syntology_url":"https://syntology.ai/paper/1903.02020","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.02020"}},"official":null}},{"url":"/paper/combinational-q-learning-for-dou-di-zhu","slug":"combinational-q-learning-for-dou-di-zhu","title":"Combinational Q-Learning for Dou Di Zhu","date":"2019-01-24","arxiv_id":"1901.08925","repositories_listed":1,"syntology":null},{"url":"/paper/information-directed-exploration-for-deep","slug":"information-directed-exploration-for-deep","title":"Information-Directed Exploration for Deep Reinforcement Learning","date":"2018-12-18","arxiv_id":"1812.07544","repositories_listed":1,"syntology":null},{"url":"/paper/an-atari-model-zoo-for-analyzing-visualizing","slug":"an-atari-model-zoo-for-analyzing-visualizing","title":"An Atari Model Zoo for Analyzing, Visualizing, and Comparing Deep Reinforcement Learning Agents","date":"2018-12-17","arxiv_id":"1812.07069","repositories_listed":1,"syntology":null},{"url":"/paper/pseudo-rehearsal-achieving-deep-reinforcement","slug":"pseudo-rehearsal-achieving-deep-reinforcement","title":"Pseudo-Rehearsal: Achieving Deep Reinforcement Learning without Catastrophic Forgetting","date":"2018-12-06","arxiv_id":"1812.02464","repositories_listed":1,"syntology":null},{"url":"/paper/adapting-auxiliary-losses-using-gradient","slug":"adapting-auxiliary-losses-using-gradient","title":"Adapting Auxiliary Losses Using Gradient Similarity","date":"2018-12-05","arxiv_id":"1812.02224","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-regularization-for-markov-decision","slug":"temporal-regularization-for-markov-decision","title":"Temporal Regularization for Markov Decision Process","date":"2018-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/efficient-eligibility-traces-for-deep","slug":"efficient-eligibility-traces-for-deep","title":"Reconciling $λ$-Returns with Experience Replay","date":"2018-10-23","arxiv_id":"1810.09967","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-perturbed-rewards","slug":"reinforcement-learning-with-perturbed-rewards","title":"Reinforcement Learning with Perturbed Rewards","date":"2018-10-02","arxiv_id":"1810.01032","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-perturbed-rewards#ran","syntology_url":"https://syntology.ai/paper/1810.01032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.01032"}},"official":{"repos":["wangjksjtu/rl-perturbed-reward"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generalization-and-regularization-in-dqn","slug":"generalization-and-regularization-in-dqn","title":"Generalization and Regularization in DQN","date":"2018-09-29","arxiv_id":"1810.00123","repositories_listed":1,"syntology":null},{"url":"/paper/transparency-and-explanation-in-deep","slug":"transparency-and-explanation-in-deep","title":"Transparency and Explanation in Deep Reinforcement Learning Neural Networks","date":"2018-09-17","arxiv_id":"1809.06061","repositories_listed":1,"syntology":null},{"url":"/paper/challenges-of-context-and-time-in","slug":"challenges-of-context-and-time-in","title":"Challenges of Context and Time in Reinforcement Learning: Introducing Space Fortress as a Benchmark","date":"2018-09-06","arxiv_id":"1809.02206","repositories_listed":1,"syntology":null},{"url":"/paper/visual-transfer-between-atari-games-using","slug":"visual-transfer-between-atari-games-using","title":"Visual Transfer between Atari Games using Competitive Reinforcement Learning","date":"2018-09-02","arxiv_id":"1809.00397","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/visual-transfer-between-atari-games-using#ran","syntology_url":"https://syntology.ai/paper/1809.00397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.00397"}},"official":{"repos":["sowmya-mp/rl_a3c_pytorch"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-option-critic-learning-safety-in-the","slug":"safe-option-critic-learning-safety-in-the","title":"Safe Option-Critic: Learning Safety in the Option-Critic Architecture","date":"2018-07-21","arxiv_id":"1807.08060","repositories_listed":1,"syntology":null},{"url":"/paper/solving-atari-games-using-fractals-and","slug":"solving-atari-games-using-fractals-and","title":"Solving Atari Games Using Fractals And Entropy","date":"2018-07-03","arxiv_id":"1807.01081","repositories_listed":1,"syntology":null},{"url":"/paper/playing-atari-with-six-neurons","slug":"playing-atari-with-six-neurons","title":"Playing Atari with Six Neurons","date":"2018-06-04","arxiv_id":"1806.01363","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-video-object-segmentation-for","slug":"unsupervised-video-object-segmentation-for","title":"Unsupervised Video Object Segmentation for Deep Reinforcement Learning","date":"2018-05-20","arxiv_id":"1805.07780","repositories_listed":1,"syntology":null},{"url":"/paper/learning-time-sensitive-strategies-in-space","slug":"learning-time-sensitive-strategies-in-space","title":"Learning Time-Sensitive Strategies in Space Fortress","date":"2018-05-17","arxiv_id":"1805.06824","repositories_listed":1,"syntology":null},{"url":"/paper/progress-compress-a-scalable-framework-for","slug":"progress-compress-a-scalable-framework-for","title":"Progress & Compress: A scalable framework for continual learning","date":"2018-05-16","arxiv_id":"1805.06370","repositories_listed":1,"syntology":null},{"url":"/paper/on-learning-intrinsic-rewards-for-policy","slug":"on-learning-intrinsic-rewards-for-policy","title":"On Learning Intrinsic Rewards for Policy Gradient Methods","date":"2018-04-17","arxiv_id":"1804.06459","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/on-learning-intrinsic-rewards-for-policy#ran","syntology_url":"https://syntology.ai/paper/1804.06459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.06459"}},"official":{"repos":["Hwhitetooth/lirpg"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/back-to-basics-benchmarking-canonical","slug":"back-to-basics-benchmarking-canonical","title":"Back to Basics: Benchmarking Canonical Evolution Strategies for Playing Atari","date":"2018-02-24","arxiv_id":"1802.08842","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-exploration-through-bayesian-deep-q","slug":"efficient-exploration-through-bayesian-deep-q","title":"Efficient Exploration through Bayesian Deep Q-Networks","date":"2018-02-13","arxiv_id":"1802.04412","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-deep-reinforcement-learning-learn","slug":"distributed-deep-reinforcement-learning-learn","title":"Distributed Deep Reinforcement Learning: Learn how to play Atari games in 21 minutes","date":"2018-01-09","arxiv_id":"1801.02852","repositories_listed":1,"syntology":null},{"url":"/paper/assumed-density-filtering-q-learning","slug":"assumed-density-filtering-q-learning","title":"Assumed Density Filtering Q-learning","date":"2017-12-09","arxiv_id":"1712.03333","repositories_listed":1,"syntology":null},{"url":"/paper/crossmodal-attentive-skill-learner","slug":"crossmodal-attentive-skill-learner","title":"Crossmodal Attentive Skill Learner","date":"2017-11-28","arxiv_id":"1711.10314","repositories_listed":1,"syntology":null},{"url":"/paper/implementing-the-deep-q-network","slug":"implementing-the-deep-q-network","title":"Implementing the Deep Q-Network","date":"2017-11-20","arxiv_id":"1711.07478","repositories_listed":1,"syntology":null},{"url":"/paper/treeqn-and-atreec-differentiable-tree","slug":"treeqn-and-atreec-differentiable-tree","title":"TreeQN and ATreeC: Differentiable Tree-Structured Models for Deep Reinforcement Learning","date":"2017-10-31","arxiv_id":"1710.11417","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/treeqn-and-atreec-differentiable-tree#ran","syntology_url":"https://syntology.ai/paper/1710.11417","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.11417"}},"official":{"repos":["oxwhirl/treeqn"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/eigenoption-discovery-through-the-deep","slug":"eigenoption-discovery-through-the-deep","title":"Eigenoption Discovery through the Deep Successor Representation","date":"2017-10-30","arxiv_id":"1710.11089","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eigenoption-discovery-through-the-deep#ran","syntology_url":"https://syntology.ai/paper/1710.11089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.11089"}},"official":null}},{"url":"/paper/generalization-tower-network-a-novel-deep","slug":"generalization-tower-network-a-novel-deep","title":"Generalization Tower Network: A Novel Deep Neural Network Architecture for Multi-Task Learning","date":"2017-10-27","arxiv_id":"1710.10036","repositories_listed":1,"syntology":null},{"url":"/paper/when-waiting-is-not-an-option-learning","slug":"when-waiting-is-not-an-option-learning","title":"When Waiting is not an Option : Learning Options with a Deliberation Cost","date":"2017-09-14","arxiv_id":"1709.04571","repositories_listed":1,"syntology":null},{"url":"/paper/trial-without-error-towards-safe","slug":"trial-without-error-towards-safe","title":"Trial without Error: Towards Safe Reinforcement Learning via Human Intervention","date":"2017-07-17","arxiv_id":"1707.05173","repositories_listed":1,"syntology":null},{"url":"/paper/count-based-exploration-in-feature-space-for","slug":"count-based-exploration-in-feature-space-for","title":"Count-Based Exploration in Feature Space for Reinforcement Learning","date":"2017-06-25","arxiv_id":"1706.08090","repositories_listed":1,"syntology":null},{"url":"/paper/on-improving-deep-reinforcement-learning-for","slug":"on-improving-deep-reinforcement-learning-for","title":"On Improving Deep Reinforcement Learning for POMDPs","date":"2017-04-26","arxiv_id":"1704.07978","repositories_listed":1,"syntology":null},{"url":"/paper/beating-atari-with-natural-language-guided","slug":"beating-atari-with-natural-language-guided","title":"Beating Atari with Natural Language Guided Reinforcement Learning","date":"2017-04-18","arxiv_id":"1704.05539","repositories_listed":1,"syntology":null}],"record_sha256":"014c91760fedb3c15c83ea359366e5550a89c5ec55a19a98f8b91a51641cb881","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}