{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/atari-games/papers/2","list_of":"/task/atari-games","task":"Atari Games","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":7,"rows_per_page":100,"rows":[101,200],"of":625,"counts":{"archive_papers_tagged":625,"with_a_code_link":313,"where_syntology_ran_a_sample":118,"not_listed_spam_title":0,"listed":625,"listed_where_code_ran":118,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":97,"every_run_a_failure_of_syntologys_instrument":21,"listed_with_a_run_with_no_instrument_failure":97,"listed_every_run_a_failure_of_syntologys_instrument":21,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/atari-games","prev":"/task/atari-games","next":"/task/atari-games/papers/3","papers":[{"url":"/paper/evolving-simple-programs-for-playing-atari","slug":"evolving-simple-programs-for-playing-atari","title":"Evolving simple programs for playing Atari games","date":"2018-06-14","arxiv_id":"1806.05695","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/evolving-simple-programs-for-playing-atari#ran","syntology_url":"https://syntology.ai/paper/1806.05695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.05695"}},"official":null}},{"url":"/paper/deep-reinforcement-learning-for-general-video","slug":"deep-reinforcement-learning-for-general-video","title":"Deep Reinforcement Learning for General Video Game AI","date":"2018-06-06","arxiv_id":"1806.02448","repositories_listed":2,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-for-general-video#ran","syntology_url":"https://syntology.ai/paper/1806.02448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.02448"}},"official":{"repos":["rubenrtorrado/GVGAI_GYM"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/spectral-inference-networks-unifying-spectral","slug":"spectral-inference-networks-unifying-spectral","title":"Spectral Inference Networks: Unifying Deep and Spectral Learning","date":"2018-06-06","arxiv_id":"1806.02215","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/spectral-inference-networks-unifying-spectral#ran","syntology_url":"https://syntology.ai/paper/1806.02215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.02215"}},"official":{"repos":["deepmind/spectral_inference_networks"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/monte-carlo-tree-search-for-asymmetric-trees","slug":"monte-carlo-tree-search-for-asymmetric-trees","title":"Monte Carlo Tree Search for Asymmetric Trees","date":"2018-05-23","arxiv_id":"1805.09218","repositories_listed":2,"syntology":null},{"url":"/paper/mean-actor-critic","slug":"mean-actor-critic","title":"Mean Actor Critic","date":"2017-09-01","arxiv_id":"1709.00503","repositories_listed":2,"syntology":null},{"url":"/paper/value-prediction-network","slug":"value-prediction-network","title":"Value Prediction Network","date":"2017-07-11","arxiv_id":"1707.03497","repositories_listed":2,"syntology":null},{"url":"/paper/elf-an-extensive-lightweight-and-flexible","slug":"elf-an-extensive-lightweight-and-flexible","title":"ELF: An Extensive, Lightweight and Flexible Research Platform for Real-time Strategy Games","date":"2017-07-04","arxiv_id":"1707.01067","repositories_listed":2,"syntology":null},{"url":"/paper/increasing-the-action-gap-new-operators-for","slug":"increasing-the-action-gap-new-operators-for","title":"Increasing the Action Gap: New Operators for Reinforcement Learning","date":"2015-12-15","arxiv_id":"1512.04860","repositories_listed":2,"syntology":null},{"url":"/paper/generalized-adaptive-transfer-network","slug":"generalized-adaptive-transfer-network","title":"Generalized Adaptive Transfer Network: Enhancing Transfer Learning in Reinforcement Learning Across Domains","date":"2025-07-02","arxiv_id":"2507.03026","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-action-duration-with-contextual","slug":"adaptive-action-duration-with-contextual","title":"Adaptive Action Duration with Contextual Bandits for Deep Reinforcement Learning in Dynamic Environments","date":"2025-06-17","arxiv_id":"2507.00030","repositories_listed":1,"syntology":null},{"url":"/paper/meta-learning-how-to-share-credit-among-macro","slug":"meta-learning-how-to-share-credit-among-macro","title":"Meta-learning how to Share Credit among Macro-Actions","date":"2025-06-16","arxiv_id":"2506.13690","repositories_listed":1,"syntology":null},{"url":"/paper/textatari-100k-frames-game-playing-with","slug":"textatari-100k-frames-game-playing-with","title":"TextAtari: 100K Frames Game Playing with Language Agents","date":"2025-06-04","arxiv_id":"2506.04098","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/textatari-100k-frames-game-playing-with#ran","syntology_url":"https://syntology.ai/paper/2506.04098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.04098"}},"official":{"repos":["Lww007/Text-Atari-Agents"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/frog-soup-zero-shot-in-context-and-sample","slug":"frog-soup-zero-shot-in-context-and-sample","title":"Frog Soup: Zero-Shot, In-Context, and Sample-Efficient Frogger Agents","date":"2025-05-06","arxiv_id":"2505.03947","repositories_listed":1,"syntology":null},{"url":"/paper/unraveling-the-rainbow-can-value-based","slug":"unraveling-the-rainbow-can-value-based","title":"Unraveling the Rainbow: can value-based methods schedule?","date":"2025-05-06","arxiv_id":"2505.03323","repositories_listed":1,"syntology":null},{"url":"/paper/pay-attention-to-what-and-where-interpretable","slug":"pay-attention-to-what-and-where-interpretable","title":"Pay Attention to What and Where? Interpretable Feature Extractor in Vision-based Deep Reinforcement Learning","date":"2025-04-14","arxiv_id":"2504.10071","repositories_listed":1,"syntology":null},{"url":"/paper/optionzero-planning-with-learned-options","slug":"optionzero-planning-with-learned-options","title":"OptionZero: Planning with Learned Options","date":"2025-02-23","arxiv_id":"2502.16634","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":7,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"7 ran (of which 7 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 7 samples that ran constructed an object rather than computing a result","sample_list":"/paper/optionzero-planning-with-learned-options#ran","syntology_url":"https://syntology.ai/paper/2502.16634","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.16634"}},"official":{"repos":["rlglab/optionzero"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":7,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/divergence-augmented-policy-optimization-1","slug":"divergence-augmented-policy-optimization-1","title":"Divergence-Augmented Policy Optimization","date":"2025-01-25","arxiv_id":"2501.15034","repositories_listed":1,"syntology":null},{"url":"/paper/utilizing-evolution-strategies-to-train","slug":"utilizing-evolution-strategies-to-train","title":"Utilizing Evolution Strategies to Train Transformers in Reinforcement Learning","date":"2025-01-23","arxiv_id":"2501.13883","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/utilizing-evolution-strategies-to-train#ran","syntology_url":"https://syntology.ai/paper/2501.13883","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.13883"}},"official":{"repos":["mafi412/evolution-strategies-and-decision-transformers"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/a-quantum-classical-reinforcement-learning","slug":"a-quantum-classical-reinforcement-learning","title":"A quantum-classical reinforcement learning model to play Atari games","date":"2024-12-11","arxiv_id":"2412.08725","repositories_listed":1,"syntology":null},{"url":"/paper/conformal-symplectic-optimization-for-stable","slug":"conformal-symplectic-optimization-for-stable","title":"Conformal Symplectic Optimization for Stable Reinforcement Learning","date":"2024-12-03","arxiv_id":"2412.02291","repositories_listed":1,"syntology":null},{"url":"/paper/decision-transformer-vs-decision-mamba","slug":"decision-transformer-vs-decision-mamba","title":"Decision Transformer vs. Decision Mamba: Analysing the Complexity of Sequential Decision Making in Atari Games","date":"2024-12-01","arxiv_id":"2412.00725","repositories_listed":1,"syntology":null},{"url":"/paper/cale-continuous-arcade-learning-environment","slug":"cale-continuous-arcade-learning-environment","title":"CALE: Continuous Arcade Learning Environment","date":"2024-10-31","arxiv_id":"2410.23810","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cale-continuous-arcade-learning-environment#ran","syntology_url":"https://syntology.ai/paper/2410.23810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.23810"}},"official":{"repos":["farama-foundation/arcade-learning-environment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-chess-reinforcement-learning-with","slug":"enhancing-chess-reinforcement-learning-with","title":"Enhancing Chess Reinforcement Learning with Graph Representation","date":"2024-10-31","arxiv_id":"2410.23753","repositories_listed":1,"syntology":null},{"url":"/paper/prosody-as-a-teaching-signal-for-agent","slug":"prosody-as-a-teaching-signal-for-agent","title":"Prosody as a Teaching Signal for Agent Learning: Exploratory Studies and Algorithmic Implications","date":"2024-10-31","arxiv_id":"2410.23554","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-two-player-performance-through","slug":"enhancing-two-player-performance-through","title":"Enhancing Two-Player Performance Through Single-Player Knowledge Transfer: An Empirical Study on Atari 2600 Games","date":"2024-10-22","arxiv_id":"2410.16653","repositories_listed":1,"syntology":null},{"url":"/paper/streaming-deep-reinforcement-learning-finally","slug":"streaming-deep-reinforcement-learning-finally","title":"Streaming Deep Reinforcement Learning Finally Works","date":"2024-10-18","arxiv_id":"2410.14606","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/streaming-deep-reinforcement-learning-finally#ran","syntology_url":"https://syntology.ai/paper/2410.14606","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14606"}},"official":{"repos":["mohmdelsayed/streaming-drl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-offline-model-based-rl-via-jointly","slug":"scaling-offline-model-based-rl-via-jointly","title":"Scaling Offline Model-Based RL via Jointly-Optimized World-Action Model Pretraining","date":"2024-10-01","arxiv_id":"2410.00564","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/scaling-offline-model-based-rl-via-jointly#ran","syntology_url":"https://syntology.ai/paper/2410.00564","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00564"}},"official":{"repos":["cjreinforce/jowa"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/perceptual-similarity-for-measuring-decision","slug":"perceptual-similarity-for-measuring-decision","title":"Perceptual Similarity for Measuring Decision-Making Style and Policy Diversity in Games","date":"2024-08-12","arxiv_id":"2408.06051","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-pre-training-objectives-for","slug":"investigating-pre-training-objectives-for","title":"Investigating Pre-Training Objectives for Generalization in Vision-Based Reinforcement Learning","date":"2024-06-10","arxiv_id":"2406.06037","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/investigating-pre-training-objectives-for#ran","syntology_url":"https://syntology.ai/paper/2406.06037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06037"}},"official":{"repos":["dojeon-ai/atari-pb"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/fdqn-a-flexible-deep-q-network-framework-for","slug":"fdqn-a-flexible-deep-q-network-framework-for","title":"FDQN: A Flexible Deep Q-Network Framework for Game Automation","date":"2024-05-29","arxiv_id":"2405.18761","repositories_listed":1,"syntology":null},{"url":"/paper/symmetric-reinforcement-learning-loss-for","slug":"symmetric-reinforcement-learning-loss-for","title":"Symmetric Reinforcement Learning Loss for Robust Learning on Diverse Tasks and Model Scales","date":"2024-05-27","arxiv_id":"2405.17618","repositories_listed":1,"syntology":null},{"url":"/paper/intelligent-go-explore-standing-on-the","slug":"intelligent-go-explore-standing-on-the","title":"Intelligent Go-Explore: Standing on the Shoulders of Giant Foundation Models","date":"2024-05-24","arxiv_id":"2405.15143","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/intelligent-go-explore-standing-on-the#ran","syntology_url":"https://syntology.ai/paper/2405.15143","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15143"}},"official":{"repos":["conglu1997/intelligent-go-explore"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/interpretable-and-editable-programmatic-tree","slug":"interpretable-and-editable-programmatic-tree","title":"Interpretable and Editable Programmatic Tree Policies for Reinforcement Learning","date":"2024-05-23","arxiv_id":"2405.14956","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-play-atari-games-using-dueling-q","slug":"learning-to-play-atari-games-using-dueling-q","title":"Learning To Play Atari Games Using Dueling Q-Learning and Hebbian Plasticity","date":"2024-05-22","arxiv_id":"2405.13960","repositories_listed":1,"syntology":null},{"url":"/paper/is-mamba-compatible-with-trajectory","slug":"is-mamba-compatible-with-trajectory","title":"Is Mamba Compatible with Trajectory Optimization in Offline Reinforcement Learning?","date":"2024-05-20","arxiv_id":"2405.12094","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/is-mamba-compatible-with-trajectory#ran","syntology_url":"https://syntology.ai/paper/2405.12094","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.12094"}},"official":{"repos":["AndssY/DeMa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/echoes-of-socratic-doubt-embracing","slug":"echoes-of-socratic-doubt-embracing","title":"Echoes of Socratic Doubt: Embracing Uncertainty in Calibrated Evidential Reinforcement Learning","date":"2024-02-11","arxiv_id":"2402.07107","repositories_listed":1,"syntology":null},{"url":"/paper/solving-deep-reinforcement-learning","slug":"solving-deep-reinforcement-learning","title":"Solving Deep Reinforcement Learning Tasks with Evolution Strategies and Linear Policy Networks","date":"2024-02-10","arxiv_id":"2402.06912","repositories_listed":1,"syntology":null},{"url":"/paper/a-robust-quantile-huber-loss-with","slug":"a-robust-quantile-huber-loss-with","title":"A Robust Quantile Huber Loss With Interpretable Parameter Adjustment In Distributional Reinforcement Learning","date":"2024-01-04","arxiv_id":"2401.02325","repositories_listed":1,"syntology":null},{"url":"/paper/distributional-bellman-operators-over-mean","slug":"distributional-bellman-operators-over-mean","title":"Distributional Bellman Operators over Mean Embeddings","date":"2023-12-09","arxiv_id":"2312.07358","repositories_listed":1,"syntology":null},{"url":"/paper/absolute-policy-optimization","slug":"absolute-policy-optimization","title":"Absolute Policy Optimization","date":"2023-10-20","arxiv_id":"2310.13230","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/absolute-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2310.13230","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.13230"}},"official":{"repos":["intelligent-control-lab/absolute-policy-optimization"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/minizero-comparative-analysis-of-alphazero","slug":"minizero-comparative-analysis-of-alphazero","title":"MiniZero: Comparative Analysis of AlphaZero and MuZero on Go, Othello, and Atari Games","date":"2023-10-17","arxiv_id":"2310.11305","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/minizero-comparative-analysis-of-alphazero#ran","syntology_url":"https://syntology.ai/paper/2310.11305","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.11305"}},"official":{"repos":["rlglab/minizero"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pgdqn-preference-guided-deep-q-network","slug":"pgdqn-preference-guided-deep-q-network","title":"PGDQN: Preference-Guided Deep Q-Network","date":"2023-10-03","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-of-generalizable-and-interpretable","slug":"learning-of-generalizable-and-interpretable","title":"Learning of Generalizable and Interpretable Knowledge in Grid-Based Reinforcement Learning Environments","date":"2023-09-07","arxiv_id":"2309.03651","repositories_listed":1,"syntology":null},{"url":"/paper/intrinsic-motivation-via-surprise-memory","slug":"intrinsic-motivation-via-surprise-memory","title":"Beyond Surprise: Improving Exploration Through Surprise Novelty","date":"2023-08-09","arxiv_id":"2308.04836","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/intrinsic-motivation-via-surprise-memory#ran","syntology_url":"https://syntology.ai/paper/2308.04836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.04836"}},"official":{"repos":["thaihungle/sm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/approximate-model-based-shielding-for-safe","slug":"approximate-model-based-shielding-for-safe","title":"Approximate Model-Based Shielding for Safe Reinforcement Learning","date":"2023-07-27","arxiv_id":"2308.00707","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-laws-for-imitation-learning-in","slug":"scaling-laws-for-imitation-learning-in","title":"Scaling Laws for Imitation Learning in Single-Agent Games","date":"2023-07-18","arxiv_id":"2307.09423","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-laws-for-imitation-learning-in#ran","syntology_url":"https://syntology.ai/paper/2307.09423","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.09423"}},"official":{"repos":["princeton-nlp/il-scaling-in-games"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-differentiable-decision-trees-learn","slug":"can-differentiable-decision-trees-learn","title":"Can Differentiable Decision Trees Enable Interpretable Reward Learning from Human Feedback?","date":"2023-06-22","arxiv_id":"2306.13004","repositories_listed":1,"syntology":null},{"url":"/paper/recurrent-memory-decision-transformer","slug":"recurrent-memory-decision-transformer","title":"Recurrent Action Transformer with Memory","date":"2023-06-15","arxiv_id":"2306.09459","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recurrent-memory-decision-transformer#ran","syntology_url":"https://syntology.ai/paper/2306.09459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09459"}},"official":{"repos":["airi-institute/rate"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ocatari-object-centric-atari-2600","slug":"ocatari-object-centric-atari-2600","title":"OCAtari: Object-Centric Atari 2600 Reinforcement Learning Environments","date":"2023-06-14","arxiv_id":"2306.08649","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ocatari-object-centric-atari-2600#ran","syntology_url":"https://syntology.ai/paper/2306.08649","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08649"}},"official":{"repos":["k4ntz/oc_atari"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/think-before-you-act-decision-transformers","slug":"think-before-you-act-decision-transformers","title":"Think Before You Act: Decision Transformers with Working Memory","date":"2023-05-24","arxiv_id":"2305.16338","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":3,"n_ran_checked":5,"n_instrument":2,"n_unverified":5,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/think-before-you-act-decision-transformers#ran","syntology_url":"https://syntology.ai/paper/2305.16338","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.16338"}},"official":{"repos":["luciferkonn/dt_mem"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/train-a-real-world-local-path-planner-in-one","slug":"train-a-real-world-local-path-planner-in-one","title":"Train a Real-world Local Path Planner in One Hour via Partially Decoupled Reinforcement Learning and Vectorized Diversity","date":"2023-05-07","arxiv_id":"2305.04180","repositories_listed":1,"syntology":null},{"url":"/paper/proto-value-networks-scaling-representation","slug":"proto-value-networks-scaling-representation","title":"Proto-Value Networks: Scaling Representation Learning with Auxiliary Tasks","date":"2023-04-25","arxiv_id":"2304.12567","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-representation-learning-in","slug":"unsupervised-representation-learning-in","title":"Unsupervised Representation Learning in Partially Observable Atari Games","date":"2023-03-13","arxiv_id":"2303.07437","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-guide-your-learner-imitation-learning","slug":"how-to-guide-your-learner-imitation-learning","title":"How To Guide Your Learner: Imitation Learning with Active Adaptive Expert Involvement","date":"2023-03-03","arxiv_id":"2303.02073","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-bellman-errors-for-offline-model","slug":"revisiting-bellman-errors-for-offline-model","title":"Revisiting Bellman Errors for Offline Model Selection","date":"2023-01-31","arxiv_id":"2302.00141","repositories_listed":1,"syntology":null},{"url":"/paper/online-real-time-recurrent-learning-using","slug":"online-real-time-recurrent-learning-using","title":"Scalable Real-Time Recurrent Learning Using Columnar-Constructive Networks","date":"2023-01-20","arxiv_id":"2302.05326","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-perceive-in-deep-model-free","slug":"learning-to-perceive-in-deep-model-free","title":"Learning to Perceive in Deep Model-Free Reinforcement Learning","date":"2023-01-10","arxiv_id":"2301.03730","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-object-representation-learning-via","slug":"boosting-object-representation-learning-via","title":"Boosting Object Representation Learning via Motion and Object Continuity","date":"2022-11-16","arxiv_id":"2211.09771","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/boosting-object-representation-learning-via#ran","syntology_url":"https://syntology.ai/paper/2211.09771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.09771"}},"official":{"repos":["k4ntz/moc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/redeeming-intrinsic-rewards-via-constrained","slug":"redeeming-intrinsic-rewards-via-constrained","title":"Redeeming Intrinsic Rewards via Constrained Optimization","date":"2022-11-14","arxiv_id":"2211.07627","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/redeeming-intrinsic-rewards-via-constrained#ran","syntology_url":"https://syntology.ai/paper/2211.07627","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.07627"}},"official":{"repos":["improbable-ai/eipo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-deep-reinforcement-learning-with-1","slug":"efficient-deep-reinforcement-learning-with-1","title":"Efficient Deep Reinforcement Learning with Predictive Processing Proximal Policy Optimization","date":"2022-11-11","arxiv_id":"2211.06236","repositories_listed":1,"syntology":null},{"url":"/paper/wild-scav-benchmarking-fps-gaming-ai-on","slug":"wild-scav-benchmarking-fps-gaming-ai-on","title":"WILD-SCAV: Benchmarking FPS Gaming AI on Unity3D-based Environments","date":"2022-10-14","arxiv_id":"2210.09026","repositories_listed":1,"syntology":null},{"url":"/paper/harfang3d-dog-fight-sandbox-a-reinforcement","slug":"harfang3d-dog-fight-sandbox-a-reinforcement","title":"Harfang3D Dog-Fight Sandbox: A Reinforcement Learning Research Platform for the Customized Control Tasks of Fighter Aircrafts","date":"2022-10-13","arxiv_id":"2210.07282","repositories_listed":1,"syntology":null},{"url":"/paper/atari-5-distilling-the-arcade-learning","slug":"atari-5-distilling-the-arcade-learning","title":"Atari-5: Distilling the Arcade Learning Environment down to Five Games","date":"2022-10-05","arxiv_id":"2210.02019","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/atari-5-distilling-the-arcade-learning#ran","syntology_url":"https://syntology.ai/paper/2210.02019","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.02019"}},"official":{"repos":["maitchison/atari-5"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pretraining-the-vision-transformer-using-self","slug":"pretraining-the-vision-transformer-using-self","title":"Pretraining the Vision Transformer using self-supervised methods for vision based Deep Reinforcement Learning","date":"2022-09-22","arxiv_id":"2209.10901","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-discrete-soft-actor-critic","slug":"revisiting-discrete-soft-actor-critic","title":"Revisiting Discrete Soft Actor-Critic","date":"2022-09-21","arxiv_id":"2209.10081","repositories_listed":1,"syntology":null},{"url":"/paper/man-multi-action-networks-learning","slug":"man-multi-action-networks-learning","title":"MAN: Multi-Action Networks Learning","date":"2022-09-19","arxiv_id":"2209.09329","repositories_listed":1,"syntology":null},{"url":"/paper/importance-prioritized-policy-distillation","slug":"importance-prioritized-policy-distillation","title":"Importance Prioritized Policy Distillation","date":"2022-08-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-swin","slug":"deep-reinforcement-learning-with-swin","title":"Deep Reinforcement Learning with Swin Transformers","date":"2022-06-30","arxiv_id":"2206.15269","repositories_listed":1,"syntology":null},{"url":"/paper/dna-proximal-policy-optimization-with-a-dual","slug":"dna-proximal-policy-optimization-with-a-dual","title":"DNA: Proximal Policy Optimization with a Dual Network Architecture","date":"2022-06-20","arxiv_id":"2206.10027","repositories_listed":1,"syntology":null},{"url":"/paper/sampling-efficient-deep-reinforcement","slug":"sampling-efficient-deep-reinforcement","title":"Sampling Efficient Deep Reinforcement Learning through Preference-Guided Stochastic Exploration","date":"2022-06-20","arxiv_id":"2206.09627","repositories_listed":1,"syntology":null},{"url":"/paper/ign-implicit-generative-networks","slug":"ign-implicit-generative-networks","title":"IGN : Implicit Generative Networks","date":"2022-06-13","arxiv_id":"2206.05860","repositories_listed":1,"syntology":null},{"url":"/paper/deep-hierarchical-planning-from-pixels","slug":"deep-hierarchical-planning-from-pixels","title":"Deep Hierarchical Planning from Pixels","date":"2022-06-08","arxiv_id":"2206.04114","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-tabula-rasa-reincarnating","slug":"beyond-tabula-rasa-reincarnating","title":"Reincarnating Reinforcement Learning: Reusing Prior Computation to Accelerate Progress","date":"2022-06-03","arxiv_id":"2206.01626","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/beyond-tabula-rasa-reincarnating#ran","syntology_url":"https://syntology.ai/paper/2206.01626","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01626"}},"official":{"repos":["google-research/reincarnating_rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/graph-backup-data-efficient-backup-exploiting","slug":"graph-backup-data-efficient-backup-exploiting","title":"Graph Backup: Data Efficient Backup Exploiting Markovian Transitions","date":"2022-05-31","arxiv_id":"2205.15824","repositories_listed":1,"syntology":null},{"url":"/paper/multi-game-decision-transformers","slug":"multi-game-decision-transformers","title":"Multi-Game Decision Transformers","date":"2022-05-30","arxiv_id":"2205.15241","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-alignment-for-history-representation","slug":"temporal-alignment-for-history-representation","title":"Temporal Alignment for History Representation in Reinforcement Learning","date":"2022-04-07","arxiv_id":"2204.03525","repositories_listed":1,"syntology":null},{"url":"/paper/learning-relational-rules-from-rewards","slug":"learning-relational-rules-from-rewards","title":"Learning Relational Rules from Rewards","date":"2022-03-25","arxiv_id":"2203.13599","repositories_listed":1,"syntology":null},{"url":"/paper/navdreams-towards-camera-only-rl-navigation","slug":"navdreams-towards-camera-only-rl-navigation","title":"NavDreams: Towards Camera-Only RL Navigation Among Humans","date":"2022-03-23","arxiv_id":"2203.12299","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-generalization-of-representations-in","slug":"on-the-generalization-of-representations-in","title":"On the Generalization of Representations in Reinforcement Learning","date":"2022-03-01","arxiv_id":"2203.00543","repositories_listed":1,"syntology":null},{"url":"/paper/expose-combining-state-based-exploration-with","slug":"expose-combining-state-based-exploration-with","title":"ExPoSe: Combining State-Based Exploration with Gradient-Based Online Search","date":"2022-02-03","arxiv_id":"2202.01461","repositories_listed":1,"syntology":null},{"url":"/paper/distributional-reinforcement-learning-via","slug":"distributional-reinforcement-learning-via","title":"Distributional Reinforcement Learning with Regularized Wasserstein Loss","date":"2022-02-01","arxiv_id":"2202.00769","repositories_listed":1,"syntology":null},{"url":"/paper/settling-the-bias-and-variance-of-meta","slug":"settling-the-bias-and-variance-of-meta","title":"A Theoretical Understanding of Gradient Bias in Meta-Reinforcement Learning","date":"2021-12-31","arxiv_id":"2112.15400","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/settling-the-bias-and-variance-of-meta#ran","syntology_url":"https://syntology.ai/paper/2112.15400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.15400"}},"official":{"repos":["Benjamin-eecs/Theoretical-GMRL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/conjugated-discrete-distributions-for","slug":"conjugated-discrete-distributions-for","title":"Conjugated Discrete Distributions for Distributional Reinforcement Learning","date":"2021-12-14","arxiv_id":"2112.07424","repositories_listed":1,"syntology":null},{"url":"/paper/continual-learning-in-environments-with","slug":"continual-learning-in-environments-with","title":"Continual Learning In Environments With Polynomial Mixing Times","date":"2021-12-13","arxiv_id":"2112.07066","repositories_listed":1,"syntology":null},{"url":"/paper/human-level-control-through-directly-trained","slug":"human-level-control-through-directly-trained","title":"Human-Level Control through Directly-Trained Deep Spiking Q-Networks","date":"2021-12-13","arxiv_id":"2201.07211","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/human-level-control-through-directly-trained#ran","syntology_url":"https://syntology.ai/paper/2201.07211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.07211"}},"official":{"repos":["aptx395/deep-spiking-q-networks"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/hybrid-self-attention-neat-a-novel","slug":"hybrid-self-attention-neat-a-novel","title":"Hybrid Self-Attention NEAT: A novel evolutionary approach to improve the NEAT algorithm","date":"2021-12-07","arxiv_id":"2112.03670","repositories_listed":1,"syntology":null},{"url":"/paper/virtual-replay-cache","slug":"virtual-replay-cache","title":"Virtual Replay Cache","date":"2021-12-06","arxiv_id":"2112.03421","repositories_listed":1,"syntology":null},{"url":"/paper/noveld-a-simple-yet-effective-exploration","slug":"noveld-a-simple-yet-effective-exploration","title":"NovelD: A Simple yet Effective Exploration Criterion","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/fast-and-data-efficient-training-of-rainbow","slug":"fast-and-data-efficient-training-of-rainbow","title":"Fast and Data-Efficient Training of Rainbow: an Experimental Study on Atari","date":"2021-11-19","arxiv_id":"2111.10247","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/fast-and-data-efficient-training-of-rainbow#ran","syntology_url":"https://syntology.ai/paper/2111.10247","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.10247"}},"official":{"repos":["schmidtdominik/rainbow"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-experience-replay-through-modeling","slug":"improving-experience-replay-through-modeling","title":"Improving Experience Replay through Modeling of Similar Transitions' Sets","date":"2021-11-12","arxiv_id":"2111.06907","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-sense-the-world-leveraging-hierarchy","slug":"how-to-sense-the-world-leveraging-hierarchy","title":"How to Sense the World: Leveraging Hierarchy in Multimodal Perception for Robust Reinforcement Learning Agents","date":"2021-10-07","arxiv_id":"2110.03608","repositories_listed":1,"syntology":null},{"url":"/paper/an-unsupervised-video-game-playstyle-metric","slug":"an-unsupervised-video-game-playstyle-metric","title":"An Unsupervised Video Game Playstyle Metric via State Discretization","date":"2021-10-03","arxiv_id":"2110.00950","repositories_listed":1,"syntology":null},{"url":"/paper/width-based-planning-and-active-learning-for","slug":"width-based-planning-and-active-learning-for","title":"Is Policy Learning Overrated?: Width-Based Planning and Active Learning for Atari","date":"2021-09-30","arxiv_id":"2109.15310","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/width-based-planning-and-active-learning-for#ran","syntology_url":"https://syntology.ai/paper/2109.15310","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.15310"}},"official":{"repos":["ibm/atari-active-learning"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/evolutionary-self-replication-as-a-mechanism","slug":"evolutionary-self-replication-as-a-mechanism","title":"Evolutionary Self-Replication as a Mechanism for Producing Artificial Intelligence","date":"2021-09-16","arxiv_id":"2109.08057","repositories_listed":1,"syntology":null},{"url":"/paper/adarl-what-where-and-how-to-adapt-in-transfer","slug":"adarl-what-where-and-how-to-adapt-in-transfer","title":"AdaRL: What, Where, and How to Adapt in Transfer Reinforcement Learning","date":"2021-07-06","arxiv_id":"2107.02729","repositories_listed":1,"syntology":{"n":7,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/adarl-what-where-and-how-to-adapt-in-transfer#ran","syntology_url":"https://syntology.ai/paper/2107.02729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.02729"}},"official":{"repos":["adaptive-rl/adarl-code"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/estimates-for-the-branching-factors-of-atari","slug":"estimates-for-the-branching-factors-of-atari","title":"Estimates for the Branching Factors of Atari Games","date":"2021-07-06","arxiv_id":"2107.02385","repositories_listed":1,"syntology":null},{"url":"/paper/ensemble-and-auxiliary-tasks-for-data","slug":"ensemble-and-auxiliary-tasks-for-data","title":"Ensemble and Auxiliary Tasks for Data-Efficient Deep Reinforcement Learning","date":"2021-07-05","arxiv_id":"2107.01904","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ensemble-and-auxiliary-tasks-for-data#ran","syntology_url":"https://syntology.ai/paper/2107.01904","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.01904"}},"official":{"repos":["NUS-LID/RENAULT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improve-agents-without-retraining-parallel","slug":"improve-agents-without-retraining-parallel","title":"Improve Agents without Retraining: Parallel Tree Search with Off-Policy Correction","date":"2021-07-04","arxiv_id":"2107.01715","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-attacks-against-deep-reinforcement","slug":"real-time-attacks-against-deep-reinforcement","title":"Real-time Adversarial Perturbations against Deep Reinforcement Learning Policies: Attacks and Defenses","date":"2021-06-16","arxiv_id":"2106.08746","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/real-time-attacks-against-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2106.08746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.08746"}},"official":{"repos":["ssg-research/ad3-action-distribution-divergence-detector"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-game-theoretic-approach-to-multi-agent","slug":"a-game-theoretic-approach-to-multi-agent","title":"A Game-Theoretic Approach to Multi-Agent Trust Region Optimization","date":"2021-06-12","arxiv_id":"2106.06828","repositories_listed":1,"syntology":null}],"record_sha256":"8f7ba69ab29afafc93974705d317883e93c9d0abb5d15ab524f8b02d8df81a25","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}