{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/3","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":135,"rows_per_page":100,"rows":[201,300],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/2","next":"/task/reinforcement-learning-2/papers/4","papers":[{"url":"/paper/whatever-does-not-kill-deep-reinforcement","slug":"whatever-does-not-kill-deep-reinforcement","title":"Whatever Does Not Kill Deep Reinforcement Learning, Makes It Stronger","date":"2017-12-23","arxiv_id":"1712.09344","repositories_listed":4,"syntology":null},{"url":"/paper/ray-a-distributed-framework-for-emerging-ai","slug":"ray-a-distributed-framework-for-emerging-ai","title":"Ray: A Distributed Framework for Emerging AI Applications","date":"2017-12-16","arxiv_id":"1712.05889","repositories_listed":4,"syntology":null},{"url":"/paper/a-deeper-look-at-experience-replay","slug":"a-deeper-look-at-experience-replay","title":"A Deeper Look at Experience Replay","date":"2017-12-04","arxiv_id":"1712.01275","repositories_listed":4,"syntology":null},{"url":"/paper/embodied-question-answering","slug":"embodied-question-answering","title":"Embodied Question Answering","date":"2017-11-30","arxiv_id":"1711.11543","repositories_listed":4,"syntology":null},{"url":"/paper/deep-reinforcement-learning-that-matters","slug":"deep-reinforcement-learning-that-matters","title":"Deep Reinforcement Learning that Matters","date":"2017-09-19","arxiv_id":"1709.06560","repositories_listed":4,"syntology":null},{"url":"/paper/leveraging-demonstrations-for-deep","slug":"leveraging-demonstrations-for-deep","title":"Leveraging Demonstrations for Deep Reinforcement Learning on Robotics Problems with Sparse Rewards","date":"2017-07-27","arxiv_id":"1707.08817","repositories_listed":4,"syntology":null},{"url":"/paper/a-multi-agent-reinforcement-learning-model-of","slug":"a-multi-agent-reinforcement-learning-model-of","title":"A multi-agent reinforcement learning model of common-pool resource appropriation","date":"2017-07-20","arxiv_id":"1707.06600","repositories_listed":4,"syntology":null},{"url":"/paper/thinking-fast-and-slow-with-deep-learning-and","slug":"thinking-fast-and-slow-with-deep-learning-and","title":"Thinking Fast and Slow with Deep Learning and Tree Search","date":"2017-05-23","arxiv_id":"1705.08439","repositories_listed":4,"syntology":null},{"url":"/paper/machine-comprehension-by-text-to-text-neural","slug":"machine-comprehension-by-text-to-text-neural","title":"Machine Comprehension by Text-to-Text Neural Question Generation","date":"2017-05-04","arxiv_id":"1705.02012","repositories_listed":4,"syntology":null},{"url":"/paper/neural-episodic-control","slug":"neural-episodic-control","title":"Neural Episodic Control","date":"2017-03-06","arxiv_id":"1703.01988","repositories_listed":4,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/neural-episodic-control#ran","syntology_url":"https://syntology.ai/paper/1703.01988","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1703.01988"}},"official":null}},{"url":"/paper/reinforcement-learning-with-deep-energy-based","slug":"reinforcement-learning-with-deep-energy-based","title":"Reinforcement Learning with Deep Energy-Based Policies","date":"2017-02-27","arxiv_id":"1702.08165","repositories_listed":4,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/reinforcement-learning-with-deep-energy-based#ran","syntology_url":"https://syntology.ai/paper/1702.08165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1702.08165"}},"official":{"repos":["haarnoja/softqlearning"],"state":"official: not harvested","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"url":"/paper/multi-agent-reinforcement-learning-in","slug":"multi-agent-reinforcement-learning-in","title":"Multi-agent Reinforcement Learning in Sequential Social Dilemmas","date":"2017-02-10","arxiv_id":"1702.03037","repositories_listed":4,"syntology":null},{"url":"/paper/hierarchical-deep-reinforcement-learning","slug":"hierarchical-deep-reinforcement-learning","title":"Hierarchical Deep Reinforcement Learning: Integrating Temporal Abstraction and Intrinsic Motivation","date":"2016-04-20","arxiv_id":"1604.06057","repositories_listed":4,"syntology":null},{"url":"/paper/multiagent-cooperation-and-competition-with","slug":"multiagent-cooperation-and-competition-with","title":"Multiagent Cooperation and Competition with Deep Reinforcement Learning","date":"2015-11-27","arxiv_id":"1511.08779","repositories_listed":4,"syntology":null},{"url":"/paper/visionreasoner-unified-visual-perception-and","slug":"visionreasoner-unified-visual-perception-and","title":"VisionReasoner: Unified Visual Perception and Reasoning via Reinforcement Learning","date":"2025-05-17","arxiv_id":"2505.12081","repositories_listed":3,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":13,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/visionreasoner-unified-visual-perception-and#ran","syntology_url":"https://syntology.ai/paper/2505.12081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12081"}},"official":{"repos":["dvlab-research/VisionReasoner","hiyouga/easyr1","dvlab-research/Seg-Zero"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ttrl-test-time-reinforcement-learning","slug":"ttrl-test-time-reinforcement-learning","title":"TTRL: Test-Time Reinforcement Learning","date":"2025-04-22","arxiv_id":"2504.16084","repositories_listed":3,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":11,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":6,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ttrl-test-time-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2504.16084","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.16084"}},"official":{"repos":["prime-rl/ttrl","tsinghuac3i/awesome-rl-reasoning-recipes"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/kimi-k1-5-scaling-reinforcement-learning-with","slug":"kimi-k1-5-scaling-reinforcement-learning-with","title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","date":"2025-01-22","arxiv_id":"2501.12599","repositories_listed":3,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/kimi-k1-5-scaling-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2501.12599","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.12599"}},"official":null}},{"url":"/paper/bricksrl-a-platform-for-democratizing","slug":"bricksrl-a-platform-for-democratizing","title":"BricksRL: A Platform for Democratizing Robotics and Reinforcement Learning Research and Education with LEGO","date":"2024-06-25","arxiv_id":"2406.17490","repositories_listed":3,"syntology":null},{"url":"/paper/rebel-reinforcement-learning-via-regressing","slug":"rebel-reinforcement-learning-via-regressing","title":"REBEL: Reinforcement Learning via Regressing Relative Rewards","date":"2024-04-25","arxiv_id":"2404.16767","repositories_listed":3,"syntology":{"n":20,"n_ran":16,"n_constructed":0,"n_ran_checked":12,"n_instrument":4,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":6,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rebel-reinforcement-learning-via-regressing#ran","syntology_url":"https://syntology.ai/paper/2404.16767","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.16767"}},"official":{"repos":["Owen-Oertell/rlcm","zhaolingao/rebel"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/evolving-reservoirs-for-meta-reinforcement","slug":"evolving-reservoirs-for-meta-reinforcement","title":"Evolving Reservoirs for Meta Reinforcement Learning","date":"2023-12-09","arxiv_id":"2312.06695","repositories_listed":3,"syntology":null},{"url":"/paper/remax-a-simple-effective-and-efficient-method","slug":"remax-a-simple-effective-and-efficient-method","title":"ReMax: A Simple, Effective, and Efficient Reinforcement Learning Method for Aligning Large Language Models","date":"2023-10-16","arxiv_id":"2310.10505","repositories_listed":3,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/remax-a-simple-effective-and-efficient-method#ran","syntology_url":"https://syntology.ai/paper/2310.10505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.10505"}},"official":{"repos":["liziniu/ReMax"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/rl4co-an-extensive-reinforcement-learning-for","slug":"rl4co-an-extensive-reinforcement-learning-for","title":"RL4CO: an Extensive Reinforcement Learning for Combinatorial Optimization Benchmark","date":"2023-06-29","arxiv_id":"2306.17100","repositories_listed":3,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rl4co-an-extensive-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2306.17100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.17100"}},"official":{"repos":["ai4co/rl4co","pytorch/rl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/datasets-and-benchmarks-for-offline-safe","slug":"datasets-and-benchmarks-for-offline-safe","title":"Datasets and Benchmarks for Offline Safe Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.09303","repositories_listed":3,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/datasets-and-benchmarks-for-offline-safe#ran","syntology_url":"https://syntology.ai/paper/2306.09303","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09303"}},"official":{"repos":["liuzuxin/dsrl","liuzuxin/fsrl","liuzuxin/osrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/video-prediction-models-as-rewards-for","slug":"video-prediction-models-as-rewards-for","title":"Video Prediction Models as Rewards for Reinforcement Learning","date":"2023-05-23","arxiv_id":"2305.14343","repositories_listed":3,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 2 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/video-prediction-models-as-rewards-for#ran","syntology_url":"https://syntology.ai/paper/2305.14343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14343"}},"official":null}},{"url":"/paper/training-diffusion-models-with-reinforcement","slug":"training-diffusion-models-with-reinforcement","title":"Training Diffusion Models with Reinforcement Learning","date":"2023-05-22","arxiv_id":"2305.13301","repositories_listed":3,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/training-diffusion-models-with-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2305.13301","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13301"}},"official":{"repos":["kvablack/ddpo-pytorch"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/revisiting-the-minimalist-approach-to-offline","slug":"revisiting-the-minimalist-approach-to-offline","title":"Revisiting the Minimalist Approach to Offline Reinforcement Learning","date":"2023-05-16","arxiv_id":"2305.09836","repositories_listed":3,"syntology":{"n":16,"n_ran":15,"n_constructed":0,"n_ran_checked":13,"n_instrument":2,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":11,"n_pointer_only":1,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 2 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/revisiting-the-minimalist-approach-to-offline#ran","syntology_url":"https://syntology.ai/paper/2305.09836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.09836"}},"official":{"repos":["dt6a/rebrac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/popgym-benchmarking-partially-observable","slug":"popgym-benchmarking-partially-observable","title":"POPGym: Benchmarking Partially Observable Reinforcement Learning","date":"2023-03-03","arxiv_id":"2303.01859","repositories_listed":3,"syntology":null},{"url":"/paper/the-dormant-neuron-phenomenon-in-deep","slug":"the-dormant-neuron-phenomenon-in-deep","title":"The Dormant Neuron Phenomenon in Deep Reinforcement Learning","date":"2023-02-24","arxiv_id":"2302.12902","repositories_listed":3,"syntology":{"n":8,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/the-dormant-neuron-phenomenon-in-deep#ran","syntology_url":"https://syntology.ai/paper/2302.12902","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12902"}},"official":{"repos":["google/dopamine"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/grounding-large-language-models-in","slug":"grounding-large-language-models-in","title":"Grounding Large Language Models in Interactive Environments with Online Reinforcement Learning","date":"2023-02-06","arxiv_id":"2302.02662","repositories_listed":3,"syntology":{"n":12,"n_ran":9,"n_constructed":2,"n_ran_checked":6,"n_instrument":3,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"9 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/grounding-large-language-models-in#ran","syntology_url":"https://syntology.ai/paper/2302.02662","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.02662"}},"official":{"repos":["clementromac/lamorel","flowersteam/grounding_llms_with_online_rl","flowersteam/lamorel"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":2,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/optimizing-prompts-for-text-to-image-1","slug":"optimizing-prompts-for-text-to-image-1","title":"Optimizing Prompts for Text-to-Image Generation","date":"2022-12-19","arxiv_id":"2212.09611","repositories_listed":3,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/optimizing-prompts-for-text-to-image-1#ran","syntology_url":"https://syntology.ai/paper/2212.09611","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.09611"}},"official":{"repos":["microsoft/lmops"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/in-context-reinforcement-learning-with","slug":"in-context-reinforcement-learning-with","title":"In-context Reinforcement Learning with Algorithm Distillation","date":"2022-10-25","arxiv_id":"2210.14215","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/in-context-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2210.14215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.14215"}},"official":null}},{"url":"/paper/model-based-offline-reinforcement-learning","slug":"model-based-offline-reinforcement-learning","title":"Model-Based Offline Reinforcement Learning with Pessimism-Modulated Dynamics Belief","date":"2022-10-13","arxiv_id":"2210.06692","repositories_listed":3,"syntology":null},{"url":"/paper/is-reinforcement-learning-not-for-natural","slug":"is-reinforcement-learning-not-for-natural","title":"Is Reinforcement Learning (Not) for Natural Language Processing: Benchmarks, Baselines, and Building Blocks for Natural Language Policy Optimization","date":"2022-10-03","arxiv_id":"2210.01241","repositories_listed":3,"syntology":null},{"url":"/paper/diffusion-policies-as-an-expressive-policy","slug":"diffusion-policies-as-an-expressive-policy","title":"Diffusion Policies as an Expressive Policy Class for Offline Reinforcement Learning","date":"2022-08-12","arxiv_id":"2208.06193","repositories_listed":3,"syntology":{"n":18,"n_ran":11,"n_constructed":5,"n_ran_checked":11,"n_instrument":0,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":10,"phrase":"11 ran (of which 5 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/diffusion-policies-as-an-expressive-policy#ran","syntology_url":"https://syntology.ai/paper/2208.06193","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.06193"}},"official":{"repos":["zhendong-wang/diffusion-policies-for-offline-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/deep-reinforcement-learning-for-multi-agent-2","slug":"deep-reinforcement-learning-for-multi-agent-2","title":"Deep Reinforcement Learning for Multi-Agent Interaction","date":"2022-08-02","arxiv_id":"2208.01769","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-turbulence","slug":"deep-reinforcement-learning-for-turbulence","title":"Deep Reinforcement Learning for Turbulence Modeling in Large Eddy Simulations","date":"2022-06-21","arxiv_id":"2206.11038","repositories_listed":3,"syntology":null},{"url":"/paper/envpool-a-highly-parallel-reinforcement","slug":"envpool-a-highly-parallel-reinforcement","title":"EnvPool: A Highly Parallel Reinforcement Learning Environment Execution Engine","date":"2022-06-21","arxiv_id":"2206.10558","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/envpool-a-highly-parallel-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2206.10558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.10558"}},"official":{"repos":["sail-sg/envpool"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["named_in_paper","official"]}}},{"url":"/paper/a-unified-approach-to-reinforcement-learning","slug":"a-unified-approach-to-reinforcement-learning","title":"A Unified Approach to Reinforcement Learning, Quantal Response Equilibria, and Two-Player Zero-Sum Games","date":"2022-06-12","arxiv_id":"2206.05825","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":3,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 3 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-unified-approach-to-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2206.05825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05825"}},"official":{"repos":["deepmind/open_spiel"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["named_in_paper"]}}},{"url":"/paper/mildly-conservative-q-learning-for-offline","slug":"mildly-conservative-q-learning-for-offline","title":"Mildly Conservative Q-Learning for Offline Reinforcement Learning","date":"2022-06-09","arxiv_id":"2206.04745","repositories_listed":3,"syntology":{"n":16,"n_ran":13,"n_constructed":6,"n_ran_checked":13,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":12,"n_pointer_only":11,"phrase":"13 ran (of which 6 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 1 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mildly-conservative-q-learning-for-offline#ran","syntology_url":"https://syntology.ai/paper/2206.04745","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.04745"}},"official":{"repos":["dmksjfl/mcq"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","listed"]}}},{"url":"/paper/supported-policy-optimization-for-offline","slug":"supported-policy-optimization-for-offline","title":"Supported Policy Optimization for Offline Reinforcement Learning","date":"2022-02-13","arxiv_id":"2202.06239","repositories_listed":3,"syntology":null},{"url":"/paper/the-shapley-value-in-machine-learning","slug":"the-shapley-value-in-machine-learning","title":"The Shapley Value in Machine Learning","date":"2022-02-11","arxiv_id":"2202.05594","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-shapley-value-in-machine-learning#ran","syntology_url":"https://syntology.ai/paper/2202.05594","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.05594"}},"official":{"repos":["benedekrozemberczki/shapley"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adversarially-trained-actor-critic-for","slug":"adversarially-trained-actor-critic-for","title":"Adversarially Trained Actor Critic for Offline Reinforcement Learning","date":"2022-02-05","arxiv_id":"2202.02446","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adversarially-trained-actor-critic-for#ran","syntology_url":"https://syntology.ai/paper/2202.02446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.02446"}},"official":{"repos":["microsoft/atac"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a-closer-look-at-advantage-filtered","slug":"a-closer-look-at-advantage-filtered","title":"A Closer Look at Advantage-Filtered Behavioral Cloning in High-Noise Datasets","date":"2021-10-10","arxiv_id":"2110.04698","repositories_listed":3,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-closer-look-at-advantage-filtered#ran","syntology_url":"https://syntology.ai/paper/2110.04698","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.04698"}},"official":{"repos":["jakegrigsby/cc-afbc","jakegrigsby/deep_control","jakegrigsby/super_sac"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/the-sensory-neuron-as-a-transformer","slug":"the-sensory-neuron-as-a-transformer","title":"The Sensory Neuron as a Transformer: Permutation-Invariant Neural Networks for Reinforcement Learning","date":"2021-09-07","arxiv_id":"2109.02869","repositories_listed":3,"syntology":{"n":8,"n_ran":8,"n_constructed":5,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 5 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-sensory-neuron-as-a-transformer#ran","syntology_url":"https://syntology.ai/paper/2109.02869","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.02869"}},"official":null}},{"url":"/paper/deep-reinforcement-learning-at-the-edge-of","slug":"deep-reinforcement-learning-at-the-edge-of","title":"Deep Reinforcement Learning at the Edge of the Statistical Precipice","date":"2021-08-30","arxiv_id":"2108.13264","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":4,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 4 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-at-the-edge-of#ran","syntology_url":"https://syntology.ai/paper/2108.13264","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.13264"}},"official":{"repos":["google-research/rliable"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/end-to-end-urban-driving-by-imitating-a","slug":"end-to-end-urban-driving-by-imitating-a","title":"End-to-End Urban Driving by Imitating a Reinforcement Learning Coach","date":"2021-08-18","arxiv_id":"2108.08265","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/end-to-end-urban-driving-by-imitating-a#ran","syntology_url":"https://syntology.ai/paper/2108.08265","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.08265"}},"official":{"repos":["zhejz/carla-roach"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/androidenv-a-reinforcement-learning-platform","slug":"androidenv-a-reinforcement-learning-platform","title":"AndroidEnv: A Reinforcement Learning Platform for Android","date":"2021-05-27","arxiv_id":"2105.13231","repositories_listed":3,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/androidenv-a-reinforcement-learning-platform#ran","syntology_url":"https://syntology.ai/paper/2105.13231","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.13231"}},"official":{"repos":["deepmind/android_env"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/feasible-actor-critic-constrained","slug":"feasible-actor-critic-constrained","title":"Feasible Actor-Critic: Constrained Reinforcement Learning for Ensuring Statewise Safety","date":"2021-05-22","arxiv_id":"2105.10682","repositories_listed":3,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/feasible-actor-critic-constrained#ran","syntology_url":"https://syntology.ai/paper/2105.10682","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.10682"}},"official":{"repos":["mahaitongdae/Feasible-Actor-Critic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-multi-agent-reinforcement-learning-for-2","slug":"deep-multi-agent-reinforcement-learning-for-2","title":"Deep Multi-agent Reinforcement Learning for Highway On-Ramp Merging in Mixed Traffic","date":"2021-05-12","arxiv_id":"2105.05701","repositories_listed":3,"syntology":null},{"url":"/paper/constructions-in-combinatorics-via-neural","slug":"constructions-in-combinatorics-via-neural","title":"Constructions in combinatorics via neural networks","date":"2021-04-29","arxiv_id":"2104.14516","repositories_listed":3,"syntology":null},{"url":"/paper/podracer-architectures-for-scalable","slug":"podracer-architectures-for-scalable","title":"Podracer architectures for scalable Reinforcement Learning","date":"2021-04-13","arxiv_id":"2104.06272","repositories_listed":3,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/podracer-architectures-for-scalable#ran","syntology_url":"https://syntology.ai/paper/2104.06272","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.06272"}},"official":null}},{"url":"/paper/integrated-decision-and-control-towards","slug":"integrated-decision-and-control-towards","title":"Integrated Decision and Control: Towards Interpretable and Computationally Efficient Driving Intelligence","date":"2021-03-18","arxiv_id":"2103.10290","repositories_listed":3,"syntology":null},{"url":"/paper/learning-to-fly-a-gym-environment-with","slug":"learning-to-fly-a-gym-environment-with","title":"Learning to Fly -- a Gym Environment with PyBullet Physics for Reinforcement Learning of Multi-agent Quadcopter Control","date":"2021-03-03","arxiv_id":"2103.02142","repositories_listed":3,"syntology":null},{"url":"/paper/near-real-world-benchmarks-for-offline","slug":"near-real-world-benchmarks-for-offline","title":"NeoRL: A Near Real-World Benchmark for Offline Reinforcement Learning","date":"2021-02-01","arxiv_id":"2102.00714","repositories_listed":3,"syntology":null},{"url":"/paper/meta-variationally-intrinsic-motivated","slug":"meta-variationally-intrinsic-motivated","title":"MetaVIM: Meta Variationally Intrinsic Motivated Reinforcement Learning for Decentralized Traffic Signal Control","date":"2021-01-04","arxiv_id":"2101.00746","repositories_listed":3,"syntology":null},{"url":"/paper/learning-fair-policies-in-decentralized","slug":"learning-fair-policies-in-decentralized","title":"Learning Fair Policies in Decentralized Cooperative Multi-Agent Reinforcement Learning","date":"2020-12-17","arxiv_id":"2012.09421","repositories_listed":3,"syntology":null},{"url":"/paper/generalization-in-reinforcement-learning-by","slug":"generalization-in-reinforcement-learning-by","title":"Generalization in Reinforcement Learning by Soft Data Augmentation","date":"2020-11-26","arxiv_id":"2011.13389","repositories_listed":3,"syntology":null},{"url":"/paper/pomo-policy-optimization-with-multiple-optima","slug":"pomo-policy-optimization-with-multiple-optima","title":"POMO: Policy Optimization with Multiple Optima for Reinforcement Learning","date":"2020-10-30","arxiv_id":"2010.16011","repositories_listed":3,"syntology":{"n":8,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/pomo-policy-optimization-with-multiple-optima#ran","syntology_url":"https://syntology.ai/paper/2010.16011","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.16011"}},"official":{"repos":["yd-kwon/POMO"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/reinforcement-learning-with-random-delays-1","slug":"reinforcement-learning-with-random-delays-1","title":"Reinforcement Learning with Random Delays","date":"2020-10-06","arxiv_id":"2010.02966","repositories_listed":3,"syntology":{"n":9,"n_ran":6,"n_constructed":2,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reinforcement-learning-with-random-delays-1#ran","syntology_url":"https://syntology.ai/paper/2010.02966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.02966"}},"official":{"repos":["rmst/rlrd"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/reward-machines-exploiting-reward-function","slug":"reward-machines-exploiting-reward-function","title":"Reward Machines: Exploiting Reward Function Structure in Reinforcement Learning","date":"2020-10-06","arxiv_id":"2010.03950","repositories_listed":3,"syntology":null},{"url":"/paper/decoupling-representation-learning-from","slug":"decoupling-representation-learning-from","title":"Decoupling Representation Learning from Reinforcement Learning","date":"2020-09-14","arxiv_id":"2009.08319","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/decoupling-representation-learning-from#ran","syntology_url":"https://syntology.ai/paper/2009.08319","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.08319"}},"official":{"repos":["astooke/rlpyt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flightmare-a-flexible-quadrotor-simulator","slug":"flightmare-a-flexible-quadrotor-simulator","title":"Flightmare: A Flexible Quadrotor Simulator","date":"2020-09-01","arxiv_id":"2009.00563","repositories_listed":3,"syntology":null},{"url":"/paper/or-gym-a-reinforcement-learning-library-for","slug":"or-gym-a-reinforcement-learning-library-for","title":"OR-Gym: A Reinforcement Learning Library for Operations Research Problems","date":"2020-08-14","arxiv_id":"2008.06319","repositories_listed":3,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/or-gym-a-reinforcement-learning-library-for#ran","syntology_url":"https://syntology.ai/paper/2008.06319","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.06319"}},"official":{"repos":["hubbs5/or-gym"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/aligning-ai-with-shared-human-values","slug":"aligning-ai-with-shared-human-values","title":"Aligning AI With Shared Human Values","date":"2020-08-05","arxiv_id":"2008.02275","repositories_listed":3,"syntology":{"n":7,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/aligning-ai-with-shared-human-values#ran","syntology_url":"https://syntology.ai/paper/2008.02275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.02275"}},"official":{"repos":["hendrycks/ethics"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/babyai-1-1","slug":"babyai-1-1","title":"BabyAI 1.1","date":"2020-07-24","arxiv_id":"2007.12770","repositories_listed":3,"syntology":{"n":9,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/babyai-1-1#ran","syntology_url":"https://syntology.ai/paper/2007.12770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.12770"}},"official":{"repos":["mila-iqia/babyai"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/monte-carlo-tree-search-as-regularized-policy","slug":"monte-carlo-tree-search-as-regularized-policy","title":"Monte-Carlo Tree Search as Regularized Policy Optimization","date":"2020-07-24","arxiv_id":"2007.12509","repositories_listed":3,"syntology":null},{"url":"/paper/implicit-distributional-reinforcement","slug":"implicit-distributional-reinforcement","title":"Implicit Distributional Reinforcement Learning","date":"2020-07-13","arxiv_id":"2007.06159","repositories_listed":3,"syntology":{"n":6,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/implicit-distributional-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2007.06159","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.06159"}},"official":{"repos":["zhougroup/IDAC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/uav-path-planning-for-wireless-data","slug":"uav-path-planning-for-wireless-data","title":"UAV Path Planning for Wireless Data Harvesting: A Deep Reinforcement Learning Approach","date":"2020-07-01","arxiv_id":"2007.00544","repositories_listed":3,"syntology":null},{"url":"/paper/adversarial-soft-advantage-fitting-imitation","slug":"adversarial-soft-advantage-fitting-imitation","title":"Adversarial Soft Advantage Fitting: Imitation Learning without Policy Optimization","date":"2020-06-23","arxiv_id":"2006.13258","repositories_listed":3,"syntology":null},{"url":"/paper/shared-experience-actor-critic-for-multi","slug":"shared-experience-actor-critic-for-multi","title":"Shared Experience Actor-Critic for Multi-Agent Reinforcement Learning","date":"2020-06-12","arxiv_id":"2006.07169","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-real","slug":"deep-reinforcement-learning-for-real","title":"Deep Reinforcement learning for real autonomous mobile robot navigation in indoor environments","date":"2020-05-28","arxiv_id":"2005.13857","repositories_listed":3,"syntology":null},{"url":"/paper/implementation-matters-in-deep-policy","slug":"implementation-matters-in-deep-policy","title":"Implementation Matters in Deep Policy Gradients: A Case Study on PPO and TRPO","date":"2020-05-25","arxiv_id":"2005.12729","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/implementation-matters-in-deep-policy#ran","syntology_url":"https://syntology.ai/paper/2005.12729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.12729"}},"official":{"repos":["MadryLab/implementation-matters"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-reinforcement-learning-tutorial","slug":"offline-reinforcement-learning-tutorial","title":"Offline Reinforcement Learning: Tutorial, Review, and Perspectives on Open Problems","date":"2020-05-04","arxiv_id":"2005.01643","repositories_listed":3,"syntology":null},{"url":"/paper/ultrasound-guided-robotic-navigation-with","slug":"ultrasound-guided-robotic-navigation-with","title":"Ultrasound-Guided Robotic Navigation with Deep Reinforcement Learning","date":"2020-03-30","arxiv_id":"2003.13321","repositories_listed":3,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ultrasound-guided-robotic-navigation-with#ran","syntology_url":"https://syntology.ai/paper/2003.13321","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.13321"}},"official":{"repos":["hhase/spinal-navigation-rl","hhase/sacrum_data-set"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-efficient-reinforcement-learning","slug":"sample-efficient-reinforcement-learning","title":"Sample Efficient Reinforcement Learning through Learning from Demonstrations in Minecraft","date":"2020-03-12","arxiv_id":"2003.06066","repositories_listed":3,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/sample-efficient-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2003.06066","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.06066"}},"official":null}},{"url":"/paper/a-deep-reinforcement-learning-algorithm-using","slug":"a-deep-reinforcement-learning-algorithm-using","title":"A Deep Reinforcement Learning Algorithm Using Dynamic Attention Model for Vehicle Routing Problems","date":"2020-02-09","arxiv_id":"2002.03282","repositories_listed":3,"syntology":null},{"url":"/paper/addressing-value-estimation-errors-in","slug":"addressing-value-estimation-errors-in","title":"Distributional Soft Actor-Critic: Off-Policy Reinforcement Learning for Addressing Value Estimation Errors","date":"2020-01-09","arxiv_id":"2001.02811","repositories_listed":3,"syntology":null},{"url":"/paper/efficient-object-detection-in-large-images","slug":"efficient-object-detection-in-large-images","title":"Efficient Object Detection in Large Images using Deep Reinforcement Learning","date":"2019-12-09","arxiv_id":"1912.03966","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-object-detection-in-large-images#ran","syntology_url":"https://syntology.ai/paper/1912.03966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.03966"}},"official":{"repos":["uzkent/EfficientObjectDetection"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/reinforcement-learning-upside-down-dont","slug":"reinforcement-learning-upside-down-dont","title":"Reinforcement Learning Upside Down: Don't Predict Rewards -- Just Map Them to Actions","date":"2019-12-05","arxiv_id":"1912.02875","repositories_listed":3,"syntology":null},{"url":"/paper/empirical-study-of-off-policy-policy","slug":"empirical-study-of-off-policy-policy","title":"Empirical Study of Off-Policy Policy Evaluation for Reinforcement Learning","date":"2019-11-15","arxiv_id":"1911.06854","repositories_listed":3,"syntology":null},{"url":"/paper/real-time-reinforcement-learning","slug":"real-time-reinforcement-learning","title":"Real-Time Reinforcement Learning","date":"2019-11-11","arxiv_id":"1911.04448","repositories_listed":3,"syntology":{"n":16,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/real-time-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1911.04448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.04448"}},"official":{"repos":["rmst/rtrl","elementai/avenue"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a-model-based-reinforcement-learning-with","slug":"a-model-based-reinforcement-learning-with","title":"Model-Based Reinforcement Learning with Adversarial Training for Online Recommendation","date":"2019-11-10","arxiv_id":"1911.03845","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-model-based-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/1911.03845","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.03845"}},"official":{"repos":["JianGuanTHU/IRecGAN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/collision-avoidance-in-pedestrian-rich","slug":"collision-avoidance-in-pedestrian-rich","title":"Collision Avoidance in Pedestrian-Rich Environments with Deep Reinforcement Learning","date":"2019-10-24","arxiv_id":"1910.11689","repositories_listed":3,"syntology":null},{"url":"/paper/generalized-inner-loop-meta-learning","slug":"generalized-inner-loop-meta-learning","title":"Generalized Inner Loop Meta-Learning","date":"2019-10-03","arxiv_id":"1910.01727","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/generalized-inner-loop-meta-learning#ran","syntology_url":"https://syntology.ai/paper/1910.01727","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.01727"}},"official":{"repos":["learnables/learn2learn","facebookresearch/higher"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/c-3po-cyclic-three-phase-optimization-for","slug":"c-3po-cyclic-three-phase-optimization-for","title":"C-3PO: Cyclic-Three-Phase Optimization for Human-Robot Motion Retargeting based on Reinforcement Learning","date":"2019-09-25","arxiv_id":"1909.11303","repositories_listed":3,"syntology":null},{"url":"/paper/emergent-tool-use-from-multi-agent","slug":"emergent-tool-use-from-multi-agent","title":"Emergent Tool Use From Multi-Agent Autocurricula","date":"2019-09-17","arxiv_id":"1909.07528","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emergent-tool-use-from-multi-agent#ran","syntology_url":"https://syntology.ai/paper/1909.07528","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.07528"}},"official":{"repos":["openai/multi-agent-emergence-environments"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/behaviour-suite-for-reinforcement-learning","slug":"behaviour-suite-for-reinforcement-learning","title":"Behaviour Suite for Reinforcement Learning","date":"2019-08-09","arxiv_id":"1908.03568","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/behaviour-suite-for-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1908.03568","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.03568"}},"official":{"repos":["deepmind/bsuite"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/boosting-soft-actor-critic-emphasizing-recent","slug":"boosting-soft-actor-critic-emphasizing-recent","title":"Boosting Soft Actor-Critic: Emphasizing Recent Experience without Forgetting the Past","date":"2019-06-10","arxiv_id":"1906.04009","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/boosting-soft-actor-critic-emphasizing-recent#ran","syntology_url":"https://syntology.ai/paper/1906.04009","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.04009"}},"official":null}},{"url":"/paper/reinforcement-learning-for-slate-based","slug":"reinforcement-learning-for-slate-based","title":"Reinforcement Learning for Slate-based Recommender Systems: A Tractable Decomposition and Practical Methodology","date":"2019-05-29","arxiv_id":"1905.12767","repositories_listed":3,"syntology":null},{"url":"/paper/maximum-entropy-regularized-multi-goal","slug":"maximum-entropy-regularized-multi-goal","title":"Maximum Entropy-Regularized Multi-Goal Reinforcement Learning","date":"2019-05-21","arxiv_id":"1905.08786","repositories_listed":3,"syntology":null},{"url":"/paper/recurrent-experience-replay-in-distributed","slug":"recurrent-experience-replay-in-distributed","title":"Recurrent Experience Replay in Distributed Reinforcement Learning","date":"2019-05-01","arxiv_id":null,"repositories_listed":3,"syntology":null},{"url":"/paper/end-to-end-robotic-reinforcement-learning","slug":"end-to-end-robotic-reinforcement-learning","title":"End-to-End Robotic Reinforcement Learning without Reward Engineering","date":"2019-04-16","arxiv_id":"1904.07854","repositories_listed":3,"syntology":null},{"url":"/paper/extrapolating-beyond-suboptimal","slug":"extrapolating-beyond-suboptimal","title":"Extrapolating Beyond Suboptimal Demonstrations via Inverse Reinforcement Learning from Observations","date":"2019-04-12","arxiv_id":"1904.06387","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/extrapolating-beyond-suboptimal#ran","syntology_url":"https://syntology.ai/paper/1904.06387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.06387"}},"official":{"repos":["hiwonjoon/ICML2019-TREX"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/holist-an-environment-for-machine-learning-of","slug":"holist-an-environment-for-machine-learning-of","title":"HOList: An Environment for Machine Learning of Higher-Order Theorem Proving","date":"2019-04-05","arxiv_id":"1904.03241","repositories_listed":3,"syntology":null},{"url":"/paper/minatar-an-atari-inspired-testbed-for-more","slug":"minatar-an-atari-inspired-testbed-for-more","title":"MinAtar: An Atari-Inspired Testbed for Thorough and Reproducible Reinforcement Learning Experiments","date":"2019-03-07","arxiv_id":"1903.03176","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/minatar-an-atari-inspired-testbed-for-more#ran","syntology_url":"https://syntology.ai/paper/1903.03176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.03176"}},"official":{"repos":["kenjyoung/MinAtar"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/hierarchical-critics-assignment-for-multi","slug":"hierarchical-critics-assignment-for-multi","title":"Reinforcement Learning from Hierarchical Critics","date":"2019-02-08","arxiv_id":"1902.03079","repositories_listed":3,"syntology":null},{"url":"/paper/pipps-flexible-model-based-policy-search","slug":"pipps-flexible-model-based-policy-search","title":"PIPPS: Flexible Model-Based Policy Search Robust to the Curse of Chaos","date":"2019-02-04","arxiv_id":"1902.01240","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-imbalanced","slug":"deep-reinforcement-learning-for-imbalanced","title":"Deep Reinforcement Learning for Imbalanced Classification","date":"2019-01-05","arxiv_id":"1901.01379","repositories_listed":3,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-for-imbalanced#ran","syntology_url":"https://syntology.ai/paper/1901.01379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.01379"}},"official":{"repos":["linenus/DRL-For-imbalanced-Classification"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/toybox-better-atari-environments-for-testing","slug":"toybox-better-atari-environments-for-testing","title":"ToyBox: Better Atari Environments for Testing Reinforcement Learning Agents","date":"2018-12-06","arxiv_id":"1812.02850","repositories_listed":3,"syntology":null},{"url":"/paper/scalable-agent-alignment-via-reward-modeling","slug":"scalable-agent-alignment-via-reward-modeling","title":"Scalable agent alignment via reward modeling: a research direction","date":"2018-11-19","arxiv_id":"1811.07871","repositories_listed":3,"syntology":null}],"record_sha256":"169b8b176cd4703cdcccd47c50130abbb18de64b0fcca2bf1c9fe149fe3ffab3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}