{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/5","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":152,"rows_per_page":100,"rows":[401,500],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/4","next":"/task/reinforcement-learning-1/papers/6","papers":[{"url":"/paper/deep-attention-recurrent-q-network","slug":"deep-attention-recurrent-q-network","title":"Deep Attention Recurrent Q-Network","date":"2015-12-05","arxiv_id":"1512.01693","repositories_listed":3,"syntology":null},{"url":"/paper/actor-mimic-deep-multitask-and-transfer","slug":"actor-mimic-deep-multitask-and-transfer","title":"Actor-Mimic: Deep Multitask and Transfer Reinforcement Learning","date":"2015-11-19","arxiv_id":"1511.06342","repositories_listed":3,"syntology":null},{"url":"/paper/active-object-localization-with-deep","slug":"active-object-localization-with-deep","title":"Active Object Localization with Deep Reinforcement Learning","date":"2015-11-18","arxiv_id":"1511.06015","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-a-natural","slug":"deep-reinforcement-learning-with-a-natural","title":"Deep Reinforcement Learning with a Natural Language Action Space","date":"2015-11-14","arxiv_id":"1511.04636","repositories_listed":3,"syntology":null},{"url":"/paper/reinforcement-learning-with-parameterized","slug":"reinforcement-learning-with-parameterized","title":"Reinforcement Learning with Parameterized Actions","date":"2015-09-05","arxiv_id":"1509.01644","repositories_listed":3,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-parameterized#ran","syntology_url":"https://syntology.ai/paper/1509.01644","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1509.01644"}},"official":null}},{"url":"/paper/giraffe-using-deep-reinforcement-learning-to","slug":"giraffe-using-deep-reinforcement-learning-to","title":"Giraffe: Using Deep Reinforcement Learning to Play Chess","date":"2015-09-04","arxiv_id":"1509.01549","repositories_listed":3,"syntology":null},{"url":"/paper/massively-parallel-methods-for-deep","slug":"massively-parallel-methods-for-deep","title":"Massively Parallel Methods for Deep Reinforcement Learning","date":"2015-07-15","arxiv_id":"1507.04296","repositories_listed":3,"syntology":null},{"url":"/paper/language-understanding-for-text-based-games","slug":"language-understanding-for-text-based-games","title":"Language Understanding for Text-based Games Using Deep Reinforcement Learning","date":"2015-06-30","arxiv_id":"1506.08941","repositories_listed":3,"syntology":null},{"url":"/paper/the-arcade-learning-environment-an-evaluation","slug":"the-arcade-learning-environment-an-evaluation","title":"The Arcade Learning Environment: An Evaluation Platform for General Agents","date":"2012-07-19","arxiv_id":"1207.4708","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-arcade-learning-environment-an-evaluation#ran","syntology_url":"https://syntology.ai/paper/1207.4708","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1207.4708"}},"official":null}},{"url":"/paper/seg-r1-segmentation-can-be-surprisingly","slug":"seg-r1-segmentation-can-be-surprisingly","title":"Seg-R1: Segmentation Can Be Surprisingly Simple with Reinforcement Learning","date":"2025-06-27","arxiv_id":"2506.22624","repositories_listed":2,"syntology":{"n":23,"n_ran":22,"n_constructed":0,"n_ran_checked":13,"n_instrument":9,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":4,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 9 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/seg-r1-segmentation-can-be-surprisingly#ran","syntology_url":"https://syntology.ai/paper/2506.22624","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.22624"}},"official":{"repos":["geshang777/Seg-R1"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/advancing-multimodal-reasoning-via","slug":"advancing-multimodal-reasoning-via","title":"Advancing Multimodal Reasoning via Reinforcement Learning with Cold Start","date":"2025-05-28","arxiv_id":"2505.22334","repositories_listed":2,"syntology":null},{"url":"/paper/unsupervised-post-training-for-multi-modal","slug":"unsupervised-post-training-for-multi-modal","title":"Unsupervised Post-Training for Multi-Modal LLM Reasoning via GRPO","date":"2025-05-28","arxiv_id":"2505.22453","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unsupervised-post-training-for-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2505.22453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.22453"}},"official":{"repos":["hiyouga/easyr1","waltonfuture/mm-upt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/r1-sharevl-incentivizing-reasoning-capability","slug":"r1-sharevl-incentivizing-reasoning-capability","title":"R1-ShareVL: Incentivizing Reasoning Capability of Multimodal Large Language Models via Share-GRPO","date":"2025-05-22","arxiv_id":"2505.16673","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/r1-sharevl-incentivizing-reasoning-capability#ran","syntology_url":"https://syntology.ai/paper/2505.16673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16673"}},"official":{"repos":["hjyao00/r1-sharevl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sophiavl-r1-reinforcing-mllms-reasoning-with","slug":"sophiavl-r1-reinforcing-mllms-reasoning-with","title":"SophiaVL-R1: Reinforcing MLLMs Reasoning with Thinking Reward","date":"2025-05-22","arxiv_id":"2505.17018","repositories_listed":2,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":9,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 9 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sophiavl-r1-reinforcing-mllms-reasoning-with#ran","syntology_url":"https://syntology.ai/paper/2505.17018","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17018"}},"official":{"repos":["hiyouga/easyr1","kxfan2002/sophiavl-r1"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-visual-grounding-for-gui-agents-via","slug":"enhancing-visual-grounding-for-gui-agents-via","title":"Enhancing Visual Grounding for GUI Agents via Self-Evolutionary Reinforcement Learning","date":"2025-05-18","arxiv_id":"2505.12370","repositories_listed":2,"syntology":null},{"url":"/paper/2505-10832","slug":"2505-10832","title":"Learning When to Think: Shaping Adaptive Reasoning in R1-Style Models via Multi-Stage RL","date":"2025-05-16","arxiv_id":"2505.10832","repositories_listed":2,"syntology":null},{"url":"/paper/agent-rl-scaling-law-agent-rl-with","slug":"agent-rl-scaling-law-agent-rl-with","title":"Agent RL Scaling Law: Agent RL with Spontaneous Code Execution for Mathematical Problem Solving","date":"2025-05-12","arxiv_id":"2505.07773","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/agent-rl-scaling-law-agent-rl-with#ran","syntology_url":"https://syntology.ai/paper/2505.07773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07773"}},"official":{"repos":["anonymize-author/agentrl","yyht/openrlhf_async_pipline"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/t2i-r1-reinforcing-image-generation-with","slug":"t2i-r1-reinforcing-image-generation-with","title":"T2I-R1: Reinforcing Image Generation with Collaborative Semantic-level and Token-level CoT","date":"2025-05-01","arxiv_id":"2505.00703","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/t2i-r1-reinforcing-image-generation-with#ran","syntology_url":"https://syntology.ai/paper/2505.00703","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.00703"}},"official":{"repos":["caraj7/t2i-r1"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/ragen-understanding-self-evolution-in-llm","slug":"ragen-understanding-self-evolution-in-llm","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","date":"2025-04-24","arxiv_id":"2504.20073","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ragen-understanding-self-evolution-in-llm#ran","syntology_url":"https://syntology.ai/paper/2504.20073","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.20073"}},"official":{"repos":["ragen-ai/ragen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-clean-slate-for-offline-reinforcement","slug":"a-clean-slate-for-offline-reinforcement","title":"A Clean Slate for Offline Reinforcement Learning","date":"2025-04-15","arxiv_id":"2504.11453","repositories_listed":2,"syntology":null},{"url":"/paper/zero-shot-whole-body-humanoid-control-via","slug":"zero-shot-whole-body-humanoid-control-via","title":"Zero-Shot Whole-Body Humanoid Control via Behavioral Foundation Models","date":"2025-04-15","arxiv_id":"2504.11054","repositories_listed":2,"syntology":{"n":31,"n_ran":16,"n_constructed":9,"n_ran_checked":10,"n_instrument":6,"n_unverified":15,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":31,"phrase":"16 ran (of which 9 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 6 where Syntology's instrument failed) · 15 unverified","sample_list":"/paper/zero-shot-whole-body-humanoid-control-via#ran","syntology_url":"https://syntology.ai/paper/2504.11054","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11054"}},"official":{"repos":["facebookresearch/humenv","facebookresearch/metamotivo"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":9,"n_ran_no_instrument_failure":10,"n_unverified":15,"ran_from_kinds":["official"]}}},{"url":"/paper/surrogate-learning-in-meta-black-box","slug":"surrogate-learning-in-meta-black-box","title":"Surrogate Learning in Meta-Black-Box Optimization: A Preliminary Study","date":"2025-03-23","arxiv_id":"2503.18060","repositories_listed":2,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/surrogate-learning-in-meta-black-box#ran","syntology_url":"https://syntology.ai/paper/2503.18060","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.18060"}},"official":{"repos":["gmc-drl/surr-rlde","GMC-DRL/MetaBox"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-outperforms-supervised","slug":"reinforcement-learning-outperforms-supervised","title":"Reinforcement Learning Outperforms Supervised Fine-Tuning: A Case Study on Audio Question Answering","date":"2025-03-14","arxiv_id":"2503.11197","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-outperforms-supervised#ran","syntology_url":"https://syntology.ai/paper/2503.11197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.11197"}},"official":{"repos":["xiaomi-research/r1-aqa","huggingface.co/mispeech/r1-aqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flow-q-learning","slug":"flow-q-learning","title":"Flow Q-Learning","date":"2025-02-04","arxiv_id":"2502.02538","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_constructed":4,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/flow-q-learning#ran","syntology_url":"https://syntology.ai/paper/2502.02538","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.02538"}},"official":{"repos":["seohongpark/fql"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/evorl-a-gpu-accelerated-framework-for","slug":"evorl-a-gpu-accelerated-framework-for","title":"EvoRL: A GPU-accelerated Framework for Evolutionary Reinforcement Learning","date":"2025-01-25","arxiv_id":"2501.15129","repositories_listed":2,"syntology":null},{"url":"/paper/offline-reinforcement-learning-for-llm-multi","slug":"offline-reinforcement-learning-for-llm-multi","title":"Offline Reinforcement Learning for LLM Multi-Step Reasoning","date":"2024-12-20","arxiv_id":"2412.16145","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/offline-reinforcement-learning-for-llm-multi#ran","syntology_url":"https://syntology.ai/paper/2412.16145","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.16145"}},"official":{"repos":["jwhj/oreo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/latent-reward-llm-empowered-credit-assignment","slug":"latent-reward-llm-empowered-credit-assignment","title":"Latent Reward: LLM-Empowered Credit Assignment in Episodic Reinforcement Learning","date":"2024-12-15","arxiv_id":"2412.11120","repositories_listed":2,"syntology":null},{"url":"/paper/towards-effective-planning-strategies-for","slug":"towards-effective-planning-strategies-for","title":"Towards Effective Planning Strategies for Dynamic Opinion Networks","date":"2024-10-18","arxiv_id":"2410.14091","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-effective-planning-strategies-for#ran","syntology_url":"https://syntology.ai/paper/2410.14091","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14091"}},"official":{"repos":["ai4society/infospread-neurips-24","BharathMuppasani/NeurIPS-Submission-2024"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-multi-step-reasoning-abilities-of","slug":"enhancing-multi-step-reasoning-abilities-of","title":"Enhancing Multi-Step Reasoning Abilities of Language Models through Direct Q-Function Optimization","date":"2024-10-11","arxiv_id":"2410.09302","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-based-model-predictive","slug":"reinforcement-learning-based-model-predictive","title":"Reinforcement Learning-based Model Predictive Control for Greenhouse Climate Control","date":"2024-09-19","arxiv_id":"2409.12789","repositories_listed":2,"syntology":null},{"url":"/paper/training-language-models-to-self-correct-via","slug":"training-language-models-to-self-correct-via","title":"Training Language Models to Self-Correct via Reinforcement Learning","date":"2024-09-19","arxiv_id":"2409.12917","repositories_listed":2,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-for-22","slug":"multi-agent-reinforcement-learning-for-22","title":"Multi-Agent Reinforcement Learning for Autonomous Driving: A Survey","date":"2024-08-19","arxiv_id":"2408.09675","repositories_listed":2,"syntology":null},{"url":"/paper/gradient-boosting-reinforcement-learning","slug":"gradient-boosting-reinforcement-learning","title":"Gradient Boosting Reinforcement Learning","date":"2024-07-11","arxiv_id":"2407.08250","repositories_listed":2,"syntology":null},{"url":"/paper/illm-tsc-integration-reinforcement-learning","slug":"illm-tsc-integration-reinforcement-learning","title":"iLLM-TSC: Integration reinforcement learning and large language model for traffic signal control policy improvement","date":"2024-07-08","arxiv_id":"2407.06025","repositories_listed":2,"syntology":null},{"url":"/paper/oralytics-reinforcement-learning-algorithm","slug":"oralytics-reinforcement-learning-algorithm","title":"Oralytics Reinforcement Learning Algorithm","date":"2024-06-19","arxiv_id":"2406.13127","repositories_listed":2,"syntology":null},{"url":"/paper/an-llm-based-recommender-system-environment","slug":"an-llm-based-recommender-system-environment","title":"SUBER: An RL Environment with Simulated Human Behavior for Recommender Systems","date":"2024-06-01","arxiv_id":"2406.01631","repositories_listed":2,"syntology":null},{"url":"/paper/dpn-decoupling-partition-and-navigation-for","slug":"dpn-decoupling-partition-and-navigation-for","title":"DPN: Decoupling Partition and Navigation for Neural Solvers of Min-max Vehicle Routing Problems","date":"2024-05-27","arxiv_id":"2405.17272","repositories_listed":2,"syntology":null},{"url":"/paper/q-value-regularized-transformer-for-offline","slug":"q-value-regularized-transformer-for-offline","title":"Q-value Regularized Transformer for Offline Reinforcement Learning","date":"2024-05-27","arxiv_id":"2405.17098","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/q-value-regularized-transformer-for-offline#ran","syntology_url":"https://syntology.ai/paper/2405.17098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17098"}},"official":null}},{"url":"/paper/which-experiences-are-influential-for-rl","slug":"which-experiences-are-influential-for-rl","title":"Which Experiences Are Influential for RL Agents? Efficiently Estimating The Influence of Experiences","date":"2024-05-23","arxiv_id":"2405.14629","repositories_listed":2,"syntology":null},{"url":"/paper/highway-graph-to-accelerate-reinforcement","slug":"highway-graph-to-accelerate-reinforcement","title":"Highway Graph to Accelerate Reinforcement Learning","date":"2024-05-20","arxiv_id":"2405.11727","repositories_listed":2,"syntology":null},{"url":"/paper/acegen-reinforcement-learning-of-generative","slug":"acegen-reinforcement-learning-of-generative","title":"ACEGEN: Reinforcement learning of generative chemical agents for drug discovery","date":"2024-05-07","arxiv_id":"2405.04657","repositories_listed":2,"syntology":null},{"url":"/paper/flagvne-a-flexible-and-generalizable","slug":"flagvne-a-flexible-and-generalizable","title":"FlagVNE: A Flexible and Generalizable Reinforcement Learning Framework for Network Resource Allocation","date":"2024-04-19","arxiv_id":"2404.12633","repositories_listed":2,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/flagvne-a-flexible-and-generalizable#ran","syntology_url":"https://syntology.ai/paper/2404.12633","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.12633"}},"official":{"repos":["GeminiLight/flag-vne"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/ev2gym-a-flexible-v2g-simulator-for-ev-smart","slug":"ev2gym-a-flexible-v2g-simulator-for-ev-smart","title":"EV2Gym: A Flexible V2G Simulator for EV Smart Charging Research and Benchmarking","date":"2024-04-02","arxiv_id":"2404.01849","repositories_listed":2,"syntology":null},{"url":"/paper/peersimgym-an-environment-for-solving-the","slug":"peersimgym-an-environment-for-solving-the","title":"PeersimGym: An Environment for Solving the Task Offloading Problem with Reinforcement Learning","date":"2024-03-26","arxiv_id":"2403.17637","repositories_listed":2,"syntology":null},{"url":"/paper/tractoracle-towards-an-anatomically-informed","slug":"tractoracle-towards-an-anatomically-informed","title":"TractOracle: towards an anatomically-informed reward function for RL-based tractography","date":"2024-03-26","arxiv_id":"2403.17845","repositories_listed":2,"syntology":null},{"url":"/paper/archer-training-language-model-agents-via","slug":"archer-training-language-model-agents-via","title":"ArCHer: Training Language Model Agents via Hierarchical Multi-Turn RL","date":"2024-02-29","arxiv_id":"2402.19446","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/archer-training-language-model-agents-via#ran","syntology_url":"https://syntology.ai/paper/2402.19446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.19446"}},"official":{"repos":["yifeizhou02/archer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/foundation-policies-with-hilbert","slug":"foundation-policies-with-hilbert","title":"Foundation Policies with Hilbert Representations","date":"2024-02-23","arxiv_id":"2402.15567","repositories_listed":2,"syntology":{"n":16,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":16,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/foundation-policies-with-hilbert#ran","syntology_url":"https://syntology.ai/paper/2402.15567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15567"}},"official":{"repos":["seohongpark/hilp"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/performative-reinforcement-learning-in","slug":"performative-reinforcement-learning-in","title":"Performative Reinforcement Learning in Gradually Shifting Environments","date":"2024-02-15","arxiv_id":"2402.09838","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/performative-reinforcement-learning-in#ran","syntology_url":"https://syntology.ai/paper/2402.09838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09838"}},"official":{"repos":["bsen/performative-rl-gradually-shifting-envs","rank-and-files/performative-rl-gradually-shifting-envs"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rewards-in-context-multi-objective-alignment","slug":"rewards-in-context-multi-objective-alignment","title":"Rewards-in-Context: Multi-objective Alignment of Foundation Models with Dynamic Preference Adjustment","date":"2024-02-15","arxiv_id":"2402.10207","repositories_listed":2,"syntology":{"n":10,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/rewards-in-context-multi-objective-alignment#ran","syntology_url":"https://syntology.ai/paper/2402.10207","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10207"}},"official":{"repos":["yangrui2015/ric"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/towards-efficient-and-exact-optimization-of","slug":"towards-efficient-and-exact-optimization-of","title":"Towards Efficient Exact Optimization of Language Model Alignment","date":"2024-02-01","arxiv_id":"2402.00856","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-efficient-and-exact-optimization-of#ran","syntology_url":"https://syntology.ai/paper/2402.00856","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.00856"}},"official":{"repos":["haozheji/exact-optimization"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-shot-reinforcement-learning-via-function","slug":"zero-shot-reinforcement-learning-via-function","title":"Zero-Shot Reinforcement Learning via Function Encoders","date":"2024-01-30","arxiv_id":"2401.17173","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/zero-shot-reinforcement-learning-via-function#ran","syntology_url":"https://syntology.ai/paper/2401.17173","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.17173"}},"official":{"repos":["anonymousresearcher5642/functionencoderrl","tyler-ingebrand/functionencoderrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dittogym-learning-to-control-soft-shape","slug":"dittogym-learning-to-control-soft-shape","title":"DittoGym: Learning to Control Soft Shape-Shifting Robots","date":"2024-01-24","arxiv_id":"2401.13231","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dittogym-learning-to-control-soft-shape#ran","syntology_url":"https://syntology.ai/paper/2401.13231","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13231"}},"official":{"repos":["suninghuang19/cfp","suninghuang19/dittogym"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/nlbac-a-neural-ordinary-differential","slug":"nlbac-a-neural-ordinary-differential","title":"Stable and Safe Human-aligned Reinforcement Learning through Neural Ordinary Differential Equations","date":"2024-01-23","arxiv_id":"2401.13148","repositories_listed":2,"syntology":null},{"url":"/paper/pdit-interleaving-perception-and-decision","slug":"pdit-interleaving-perception-and-decision","title":"PDiT: Interleaving Perception and Decision-making Transformers for Deep Reinforcement Learning","date":"2023-12-26","arxiv_id":"2312.15863","repositories_listed":2,"syntology":null},{"url":"/paper/optimizing-heat-alert-issuance-for-public","slug":"optimizing-heat-alert-issuance-for-public","title":"Optimizing Heat Alert Issuance with Reinforcement Learning","date":"2023-12-21","arxiv_id":"2312.14196","repositories_listed":2,"syntology":null},{"url":"/paper/active-reinforcement-learning-for-robust","slug":"active-reinforcement-learning-for-robust","title":"Active Reinforcement Learning for Robust Building Control","date":"2023-12-16","arxiv_id":"2312.10289","repositories_listed":2,"syntology":null},{"url":"/paper/learning-to-fly-in-seconds","slug":"learning-to-fly-in-seconds","title":"Learning to Fly in Seconds","date":"2023-11-22","arxiv_id":"2311.13081","repositories_listed":2,"syntology":null},{"url":"/paper/drm-mastering-visual-reinforcement-learning","slug":"drm-mastering-visual-reinforcement-learning","title":"DrM: Mastering Visual Reinforcement Learning through Dormant Ratio Minimization","date":"2023-10-30","arxiv_id":"2310.19668","repositories_listed":2,"syntology":{"n":14,"n_ran":10,"n_constructed":3,"n_ran_checked":8,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":6,"phrase":"10 ran (of which 3 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/drm-mastering-visual-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2310.19668","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19668"}},"official":{"repos":["XuGW-Kevin/DrM"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/automatic-unit-test-data-generation-and-actor","slug":"automatic-unit-test-data-generation-and-actor","title":"Automatic Unit Test Data Generation and Actor-Critic Reinforcement Learning for Code Synthesis","date":"2023-10-20","arxiv_id":"2310.13669","repositories_listed":2,"syntology":null},{"url":"/paper/towards-robust-offline-reinforcement-learning","slug":"towards-robust-offline-reinforcement-learning","title":"Towards Robust Offline Reinforcement Learning under Diverse Data Corruption","date":"2023-10-19","arxiv_id":"2310.12955","repositories_listed":2,"syntology":{"n":9,"n_ran":5,"n_constructed":2,"n_ran_checked":3,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":9,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/towards-robust-offline-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2310.12955","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12955"}},"official":{"repos":["yangrui2015/riql"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/dsac-t-distributional-soft-actor-critic-with","slug":"dsac-t-distributional-soft-actor-critic-with","title":"Distributional Soft Actor-Critic with Three Refinements","date":"2023-10-09","arxiv_id":"2310.05858","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dsac-t-distributional-soft-actor-critic-with#ran","syntology_url":"https://syntology.ai/paper/2310.05858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05858"}},"official":{"repos":["jingliang-duan/dsac-t","jingliang-duan/dsac-v2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rllte-long-term-evolution-project-of","slug":"rllte-long-term-evolution-project-of","title":"RLLTE: Long-Term Evolution Project of Reinforcement Learning","date":"2023-09-28","arxiv_id":"2309.16382","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rllte-long-term-evolution-project-of#ran","syntology_url":"https://syntology.ai/paper/2309.16382","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16382"}},"official":{"repos":["RLE-Foundation/rllte"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-toolkit-for-reliable-benchmarking-and","slug":"a-toolkit-for-reliable-benchmarking-and","title":"A Toolkit for Reliable Benchmarking and Research in Multi-Objective Reinforcement Learning","date":"2023-09-26","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/conservative-world-models","slug":"conservative-world-models","title":"Zero-Shot Reinforcement Learning from Low Quality Data","date":"2023-09-26","arxiv_id":"2309.15178","repositories_listed":2,"syntology":{"n":18,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":8,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/conservative-world-models#ran","syntology_url":"https://syntology.ai/paper/2309.15178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.15178"}},"official":{"repos":["enjeeneer/conservative-world-models","enjeeneer/zero-shot-rl"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-data-efficiency-in-reinforcement","slug":"enhancing-data-efficiency-in-reinforcement","title":"Enhancing data efficiency in reinforcement learning: a novel imagination mechanism based on mesh information propagation","date":"2023-09-25","arxiv_id":"2309.14243","repositories_listed":2,"syntology":null},{"url":"/paper/aligning-language-models-with-offline","slug":"aligning-language-models-with-offline","title":"Aligning Language Models with Offline Learning from Human Feedback","date":"2023-08-23","arxiv_id":"2308.12050","repositories_listed":2,"syntology":null},{"url":"/paper/esrl-efficient-sampling-based-reinforcement","slug":"esrl-efficient-sampling-based-reinforcement","title":"ESRL: Efficient Sampling-based Reinforcement Learning for Sequence Generation","date":"2023-08-04","arxiv_id":"2308.02223","repositories_listed":2,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/esrl-efficient-sampling-based-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2308.02223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.02223"}},"official":{"repos":["wangclnlp/DeepSpeed-Chat-Extension"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ixdrl-a-novel-explainable-deep-reinforcement","slug":"ixdrl-a-novel-explainable-deep-reinforcement","title":"IxDRL: A Novel Explainable Deep Reinforcement Learning Toolkit based on Analyses of Interestingness","date":"2023-07-18","arxiv_id":"2307.08933","repositories_listed":2,"syntology":null},{"url":"/paper/pid-inspired-inductive-biases-for-deep-1","slug":"pid-inspired-inductive-biases-for-deep-1","title":"PID-Inspired Inductive Biases for Deep Reinforcement Learning in Partially Observable Control Tasks","date":"2023-07-12","arxiv_id":"2307.05891","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":1,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pid-inspired-inductive-biases-for-deep-1#ran","syntology_url":"https://syntology.ai/paper/2307.05891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.05891"}},"official":null}},{"url":"/paper/alleviating-matthew-effect-of-offline","slug":"alleviating-matthew-effect-of-offline","title":"Alleviating Matthew Effect of Offline Reinforcement Learning in Interactive Recommendation","date":"2023-07-10","arxiv_id":"2307.04571","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/alleviating-matthew-effect-of-offline#ran","syntology_url":"https://syntology.ai/paper/2307.04571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.04571"}},"official":{"repos":["chongminggao/dorl-codes"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/when-do-transformers-shine-in-rl-decoupling-1","slug":"when-do-transformers-shine-in-rl-decoupling-1","title":"When Do Transformers Shine in RL? Decoupling Memory from Credit Assignment","date":"2023-07-07","arxiv_id":"2307.03864","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/when-do-transformers-shine-in-rl-decoupling-1#ran","syntology_url":"https://syntology.ai/paper/2307.03864","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.03864"}},"official":{"repos":["twni2016/memory-rl","twni2016/pomdp-baselines"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-bellman-inconsistency-for-model-based","slug":"model-bellman-inconsistency-for-model-based","title":"Model-Bellman Inconsistency for Model-based Offline Reinforcement Learning","date":"2023-07-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/srl-scaling-distributed-reinforcement","slug":"srl-scaling-distributed-reinforcement","title":"SRL: Scaling Distributed Reinforcement Learning to Over Ten Thousand Cores","date":"2023-06-29","arxiv_id":"2306.16688","repositories_listed":2,"syntology":null},{"url":"/paper/intercode-standardizing-and-benchmarking","slug":"intercode-standardizing-and-benchmarking","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","date":"2023-06-26","arxiv_id":"2306.14898","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/intercode-standardizing-and-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2306.14898","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.14898"}},"official":{"repos":["princeton-nlp/intercode"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unified-off-policy-learning-to-rank-a","slug":"unified-off-policy-learning-to-rank-a","title":"Unified Off-Policy Learning to Rank: a Reinforcement Learning Perspective","date":"2023-06-13","arxiv_id":"2306.07528","repositories_listed":2,"syntology":null},{"url":"/paper/policy-regularization-with-dataset-constraint","slug":"policy-regularization-with-dataset-constraint","title":"Policy Regularization with Dataset Constraint for Offline Reinforcement Learning","date":"2023-06-11","arxiv_id":"2306.06569","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":6,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 6 samples that ran constructed an object rather than computing a result","sample_list":"/paper/policy-regularization-with-dataset-constraint#ran","syntology_url":"https://syntology.ai/paper/2306.06569","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.06569"}},"official":{"repos":["lamda-rl/prdc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/offline-prioritized-experience-replay","slug":"offline-prioritized-experience-replay","title":"Decoupled Prioritized Resampling for Offline RL","date":"2023-06-08","arxiv_id":"2306.05412","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/offline-prioritized-experience-replay#ran","syntology_url":"https://syntology.ai/paper/2306.05412","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.05412"}},"official":{"repos":["sail-sg/oper","yueyang130/odpr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/for-sale-state-action-representation-learning-1","slug":"for-sale-state-action-representation-learning-1","title":"For SALE: State-Action Representation Learning for Deep Reinforcement Learning","date":"2023-06-04","arxiv_id":"2306.02451","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/for-sale-state-action-representation-learning-1#ran","syntology_url":"https://syntology.ai/paper/2306.02451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.02451"}},"official":{"repos":["sfujim/td7"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/torchrl-a-data-driven-decision-making-library","slug":"torchrl-a-data-driven-decision-making-library","title":"TorchRL: A data-driven decision-making library for PyTorch","date":"2023-06-01","arxiv_id":"2306.00577","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/torchrl-a-data-driven-decision-making-library#ran","syntology_url":"https://syntology.ai/paper/2306.00577","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00577"}},"official":null}},{"url":"/paper/dpok-reinforcement-learning-for-fine-tuning","slug":"dpok-reinforcement-learning-for-fine-tuning","title":"DPOK: Reinforcement Learning for Fine-tuning Text-to-Image Diffusion Models","date":"2023-05-25","arxiv_id":"2305.16381","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/dpok-reinforcement-learning-for-fine-tuning#ran","syntology_url":"https://syntology.ai/paper/2305.16381","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.16381"}},"official":{"repos":["google-research/google-research"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/end-to-end-meta-bayesian-optimisation-with","slug":"end-to-end-meta-bayesian-optimisation-with","title":"End-to-End Meta-Bayesian Optimisation with Transformer Neural Processes","date":"2023-05-25","arxiv_id":"2305.15930","repositories_listed":2,"syntology":null},{"url":"/paper/policy-gradient-methods-in-the-presence-of","slug":"policy-gradient-methods-in-the-presence-of","title":"Policy Gradient Methods in the Presence of Symmetries and State Abstractions","date":"2023-05-09","arxiv_id":"2305.05666","repositories_listed":2,"syntology":null},{"url":"/paper/explaining-rl-decisions-with-trajectories","slug":"explaining-rl-decisions-with-trajectories","title":"Explaining RL Decisions with Trajectories","date":"2023-05-06","arxiv_id":"2305.04073","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/explaining-rl-decisions-with-trajectories#ran","syntology_url":"https://syntology.ai/paper/2305.04073","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.04073"}},"official":{"repos":["shripaddeshmukh/xrl_with_trajectories"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/leveraging-factored-action-spaces-for","slug":"leveraging-factored-action-spaces-for","title":"Leveraging Factored Action Spaces for Efficient Offline Reinforcement Learning in Healthcare","date":"2023-05-02","arxiv_id":"2305.01738","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leveraging-factored-action-spaces-for#ran","syntology_url":"https://syntology.ai/paper/2305.01738","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.01738"}},"official":{"repos":["mld3/offlinerl_factoredactions"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/diffmimic-efficient-motion-mimicking-with","slug":"diffmimic-efficient-motion-mimicking-with","title":"DiffMimic: Efficient Motion Mimicking with Differentiable Physics","date":"2023-04-06","arxiv_id":"2304.03274","repositories_listed":2,"syntology":null},{"url":"/paper/multi-view-tensor-graph-neural-networks","slug":"multi-view-tensor-graph-neural-networks","title":"Multi-view Tensor Graph Neural Networks Through Reinforced Aggregation","date":"2023-04-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/cflownets-continuous-control-with-generative","slug":"cflownets-continuous-control-with-generative","title":"CFlowNets: Continuous Control with Generative Flow Networks","date":"2023-03-04","arxiv_id":"2303.02430","repositories_listed":2,"syntology":null},{"url":"/paper/learning-to-control-autonomous-fleets-from","slug":"learning-to-control-autonomous-fleets-from","title":"Learning to Control Autonomous Fleets from Observation via Offline Reinforcement Learning","date":"2023-02-28","arxiv_id":"2302.14833","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-control-autonomous-fleets-from#ran","syntology_url":"https://syntology.ai/paper/2302.14833","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.14833"}},"official":{"repos":["carolinssc/offline-rl-amod","carolinssc/offline-rl-for-amod"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/ganterfactual-rl-understanding-reinforcement","slug":"ganterfactual-rl-understanding-reinforcement","title":"GANterfactual-RL: Understanding Reinforcement Learning Agents' Strategies through Visual Counterfactual Explanations","date":"2023-02-24","arxiv_id":"2302.12689","repositories_listed":2,"syntology":null},{"url":"/paper/neural-laplace-control-for-continuous-time","slug":"neural-laplace-control-for-continuous-time","title":"Neural Laplace Control for Continuous-time Delayed Systems","date":"2023-02-24","arxiv_id":"2302.12604","repositories_listed":2,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":9,"n_instrument":4,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/neural-laplace-control-for-continuous-time#ran","syntology_url":"https://syntology.ai/paper/2302.12604","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12604"}},"official":{"repos":["samholt/neurallaplacecontrol","vanderschaarlab/neurallaplacecontrol"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/combining-search-strategies-to-improve","slug":"combining-search-strategies-to-improve","title":"Reinforcement Learning for Combining Search Methods in the Calibration of Economic ABMs","date":"2023-02-23","arxiv_id":"2302.11835","repositories_listed":2,"syntology":null},{"url":"/paper/behavior-proximal-policy-optimization","slug":"behavior-proximal-policy-optimization","title":"Behavior Proximal Policy Optimization","date":"2023-02-22","arxiv_id":"2302.11312","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/behavior-proximal-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2302.11312","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.11312"}},"official":{"repos":["dragon-zhuang/bppo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploration-by-self-supervised-exploitation","slug":"exploration-by-self-supervised-exploitation","title":"Self-supervised network distillation: an effective approach to exploration in sparse reward environments","date":"2023-02-22","arxiv_id":"2302.11563","repositories_listed":2,"syntology":null},{"url":"/paper/demonstration-guided-reinforcement-learning-1","slug":"demonstration-guided-reinforcement-learning-1","title":"Demonstration-Guided Reinforcement Learning with Efficient Exploration for Task Automation of Surgical Robot","date":"2023-02-20","arxiv_id":"2302.09772","repositories_listed":2,"syntology":null},{"url":"/paper/fantastic-rewards-and-how-to-tame-them-a-case","slug":"fantastic-rewards-and-how-to-tame-them-a-case","title":"Fantastic Rewards and How to Tame Them: A Case Study on Reward Learning for Task-oriented Dialogue Systems","date":"2023-02-20","arxiv_id":"2302.10342","repositories_listed":2,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/fantastic-rewards-and-how-to-tame-them-a-case#ran","syntology_url":"https://syntology.ai/paper/2302.10342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.10342"}},"official":{"repos":["budzianowski/multiwoz","shentao-yang/fantastic_reward_iclr2023"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/predictable-mdp-abstraction-for-unsupervised","slug":"predictable-mdp-abstraction-for-unsupervised","title":"Predictable MDP Abstraction for Unsupervised Model-Based RL","date":"2023-02-08","arxiv_id":"2302.03921","repositories_listed":2,"syntology":null},{"url":"/paper/efficient-online-reinforcement-learning-with","slug":"efficient-online-reinforcement-learning-with","title":"Efficient Online Reinforcement Learning with Offline Data","date":"2023-02-06","arxiv_id":"2302.02948","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-online-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2302.02948","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.02948"}},"official":{"repos":["ikostrikov/rlpd"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/locally-constrained-policy-optimization-for","slug":"locally-constrained-policy-optimization-for","title":"Online Reinforcement Learning in Non-Stationary Context-Driven Environments","date":"2023-02-04","arxiv_id":"2302.02182","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/locally-constrained-policy-optimization-for#ran","syntology_url":"https://syntology.ai/paper/2302.02182","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.02182"}},"official":{"repos":["lcpo-rl/lcpo","pouyahmdn/lcpo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/off-the-grid-marl-a-framework-for-dataset","slug":"off-the-grid-marl-a-framework-for-dataset","title":"Off-the-Grid MARL: Datasets with Baselines for Offline Multi-Agent Reinforcement Learning","date":"2023-02-01","arxiv_id":"2302.00521","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/off-the-grid-marl-a-framework-for-dataset#ran","syntology_url":"https://syntology.ai/paper/2302.00521","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00521"}},"official":{"repos":["instadeepai/og-marl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/modeling-moral-choices-in-social-dilemmas","slug":"modeling-moral-choices-in-social-dilemmas","title":"Modeling Moral Choices in Social Dilemmas with Multi-Agent Reinforcement Learning","date":"2023-01-20","arxiv_id":"2301.08491","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/modeling-moral-choices-in-social-dilemmas#ran","syntology_url":"https://syntology.ai/paper/2301.08491","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.08491"}},"official":{"repos":["liza-tennant/moral_choice_dyadic","Liza-Tennant/modeling_moral_choice_dyadic"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"0139bc256f3bf4089835c76cde5b27c82113d181aa37a3b0be2c11fc9d20c9f1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}