{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/ran/1","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":12,"rows_per_page":100,"rows":[1,100],"of":1175,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning/papers/ran/1","prev":null,"next":"/task/reinforcement-learning/papers/ran/2","papers":[{"url":"/paper/rlver-reinforcement-learning-with-verifiable","slug":"rlver-reinforcement-learning-with-verifiable","title":"RLVER: Reinforcement Learning with Verifiable Emotion Rewards for Empathetic Agents","date":"2025-07-03","arxiv_id":"2507.03112","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rlver-reinforcement-learning-with-verifiable#ran","syntology_url":"https://syntology.ai/paper/2507.03112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.03112"}},"official":{"repos":["tencent/digitalhuman"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/octothinker-mid-training-incentivizes","slug":"octothinker-mid-training-incentivizes","title":"OctoThinker: Mid-training Incentivizes Reinforcement Learning Scaling","date":"2025-06-25","arxiv_id":"2506.20512","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":4,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/octothinker-mid-training-incentivizes#ran","syntology_url":"https://syntology.ai/paper/2506.20512","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.20512"}},"official":{"repos":["gair-nlp/octothinker"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dr-sac-distributionally-robust-soft-actor","slug":"dr-sac-distributionally-robust-soft-actor","title":"DR-SAC: Distributionally Robust Soft Actor-Critic for Reinforcement Learning under Uncertainty","date":"2025-06-14","arxiv_id":"2506.12622","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/dr-sac-distributionally-robust-soft-actor#ran","syntology_url":"https://syntology.ai/paper/2506.12622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.12622"}},"official":{"repos":["lemutisme/dr-sac"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/treerl-llm-reinforcement-learning-with-on","slug":"treerl-llm-reinforcement-learning-with-on","title":"TreeRL: LLM Reinforcement Learning with On-Policy Tree Search","date":"2025-06-13","arxiv_id":"2506.11902","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/treerl-llm-reinforcement-learning-with-on#ran","syntology_url":"https://syntology.ai/paper/2506.11902","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.11902"}},"official":{"repos":["thudm/treerl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/verif-verification-engineering-for","slug":"verif-verification-engineering-for","title":"VerIF: Verification Engineering for Reinforcement Learning in Instruction Following","date":"2025-06-11","arxiv_id":"2506.09942","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/verif-verification-engineering-for#ran","syntology_url":"https://syntology.ai/paper/2506.09942","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09942"}},"official":{"repos":["thu-keg/verif"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-teachers-of-test-time","slug":"reinforcement-learning-teachers-of-test-time","title":"Reinforcement Learning Teachers of Test Time Scaling","date":"2025-06-10","arxiv_id":"2506.08388","repositories_listed":0,"syntology":{"n":25,"n_ran":18,"n_constructed":0,"n_ran_checked":18,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":18,"n_pointer_only":0,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 0 violated, 18 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/reinforcement-learning-teachers-of-test-time#ran","syntology_url":"https://syntology.ai/paper/2506.08388","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08388"}},"official":null}},{"url":"/paper/2506-08460","slug":"2506-08460","title":"MOBODY: Model Based Off-Dynamics Offline Reinforcement Learning","date":"2025-06-10","arxiv_id":"2506.08460","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":3,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/2506-08460#ran","syntology_url":"https://syntology.ai/paper/2506.08460","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08460"}},"official":{"repos":["guoyihonggyh/mobody-model-based-off-dynamics-offline-reinforcement-learning"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/co-evolving-llm-coder-and-unit-tester-via","slug":"co-evolving-llm-coder-and-unit-tester-via","title":"Co-Evolving LLM Coder and Unit Tester via Reinforcement Learning","date":"2025-06-03","arxiv_id":"2506.03136","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/co-evolving-llm-coder-and-unit-tester-via#ran","syntology_url":"https://syntology.ai/paper/2506.03136","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.03136"}},"official":{"repos":["gen-verse/cure"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/reinforcement-learning-tuning-for-videollms","slug":"reinforcement-learning-tuning-for-videollms","title":"Reinforcement Learning Tuning for VideoLLMs: Reward Design and Data Efficiency","date":"2025-06-02","arxiv_id":"2506.01908","repositories_listed":1,"syntology":{"n":17,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/reinforcement-learning-tuning-for-videollms#ran","syntology_url":"https://syntology.ai/paper/2506.01908","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.01908"}},"official":{"repos":["appletea233/temporal-r1"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/clarify-contrastive-preference-reinforcement","slug":"clarify-contrastive-preference-reinforcement","title":"CLARIFY: Contrastive Preference Reinforcement Learning for Untangling Ambiguous Queries","date":"2025-05-31","arxiv_id":"2506.00388","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":1,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":6,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clarify-contrastive-preference-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2506.00388","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.00388"}},"official":{"repos":["moonoutcloudback/clarify_pbrl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-reward-fairness-in-rlhf-from-a","slug":"towards-reward-fairness-in-rlhf-from-a","title":"Towards Reward Fairness in RLHF: From a Resource Allocation Perspective","date":"2025-05-29","arxiv_id":"2505.23349","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-reward-fairness-in-rlhf-from-a#ran","syntology_url":"https://syntology.ai/paper/2505.23349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23349"}},"official":{"repos":["shoyua/towards-reward-fairness"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cadrille-multi-modal-cad-reconstruction-with","slug":"cadrille-multi-modal-cad-reconstruction-with","title":"cadrille: Multi-modal CAD Reconstruction with Online Reinforcement Learning","date":"2025-05-28","arxiv_id":"2505.22914","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/cadrille-multi-modal-cad-reconstruction-with#ran","syntology_url":"https://syntology.ai/paper/2505.22914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.22914"}},"official":null}},{"url":"/paper/rearank-reasoning-re-ranking-agent-via","slug":"rearank-reasoning-re-ranking-agent-via","title":"REARANK: Reasoning Re-ranking Agent via Reinforcement Learning","date":"2025-05-26","arxiv_id":"2505.20046","repositories_listed":1,"syntology":{"n":18,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/rearank-reasoning-re-ranking-agent-via#ran","syntology_url":"https://syntology.ai/paper/2505.20046","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20046"}},"official":{"repos":["lezhang7/rearank"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/behavior-injection-preparing-language-models","slug":"behavior-injection-preparing-language-models","title":"Behavior Injection: Preparing Language Models for Reinforcement Learning","date":"2025-05-25","arxiv_id":"2505.18917","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/behavior-injection-preparing-language-models#ran","syntology_url":"https://syntology.ai/paper/2505.18917","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18917"}},"official":{"repos":["czp16/bridge-llm-reasoning"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/serl-self-play-reinforcement-learning-for","slug":"serl-self-play-reinforcement-learning-for","title":"SeRL: Self-Play Reinforcement Learning for Large Language Models with Limited Data","date":"2025-05-25","arxiv_id":"2505.20347","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/serl-self-play-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2505.20347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20347"}},"official":{"repos":["wantbook-book/serl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/co-reinforcement-learning-for-unified","slug":"co-reinforcement-learning-for-unified","title":"Co-Reinforcement Learning for Unified Multimodal Understanding and Generation","date":"2025-05-23","arxiv_id":"2505.17534","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/co-reinforcement-learning-for-unified#ran","syntology_url":"https://syntology.ai/paper/2505.17534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17534"}},"official":{"repos":["mm-vl/ulm-r1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/arpo-end-to-end-policy-optimization-for-gui","slug":"arpo-end-to-end-policy-optimization-for-gui","title":"ARPO:End-to-End Policy Optimization for GUI Agents with Experience Replay","date":"2025-05-22","arxiv_id":"2505.16282","repositories_listed":1,"syntology":{"n":16,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/arpo-end-to-end-policy-optimization-for-gui#ran","syntology_url":"https://syntology.ai/paper/2505.16282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16282"}},"official":{"repos":["dvlab-research/arpo"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/maximum-total-correlation-reinforcement","slug":"maximum-total-correlation-reinforcement","title":"Maximum Total Correlation Reinforcement Learning","date":"2025-05-22","arxiv_id":"2505.16734","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":6,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"8 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/maximum-total-correlation-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2505.16734","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16734"}},"official":{"repos":["bangyou01/mtc"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":6,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/got-r1-unleashing-reasoning-capability-of","slug":"got-r1-unleashing-reasoning-capability-of","title":"GoT-R1: Unleashing Reasoning Capability of MLLM for Visual Generation with Reinforcement Learning","date":"2025-05-22","arxiv_id":"2505.17022","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":6,"n_ran_checked":7,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"9 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/got-r1-unleashing-reasoning-capability-of#ran","syntology_url":"https://syntology.ai/paper/2505.17022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17022"}},"official":{"repos":["gogoduan/got-r1"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":6,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/conciserl-conciseness-guided-reinforcement","slug":"conciserl-conciseness-guided-reinforcement","title":"ConciseRL: Conciseness-Guided Reinforcement Learning for Efficient Reasoning Models","date":"2025-05-22","arxiv_id":"2505.17250","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conciserl-conciseness-guided-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2505.17250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17250"}},"official":{"repos":["razvandu/conciserl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rlvr-world-training-world-models-with","slug":"rlvr-world-training-world-models-with","title":"RLVR-World: Training World Models with Reinforcement Learning","date":"2025-05-20","arxiv_id":"2505.13934","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rlvr-world-training-world-models-with#ran","syntology_url":"https://syntology.ai/paper/2505.13934","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13934"}},"official":{"repos":["thuml/RLVR-World"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/deepeyes-incentivizing-thinking-with-images","slug":"deepeyes-incentivizing-thinking-with-images","title":"DeepEyes: Incentivizing \"Thinking with Images\" via Reinforcement Learning","date":"2025-05-20","arxiv_id":"2505.14362","repositories_listed":1,"syntology":{"n":17,"n_ran":12,"n_constructed":0,"n_ran_checked":9,"n_instrument":3,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/deepeyes-incentivizing-thinking-with-images#ran","syntology_url":"https://syntology.ai/paper/2505.14362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14362"}},"official":{"repos":["visual-agent/deepeyes"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/dual-agent-reinforcement-learning-for","slug":"dual-agent-reinforcement-learning-for","title":"Dual-Agent Reinforcement Learning for Automated Feature Generation","date":"2025-05-19","arxiv_id":"2505.12628","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":3,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dual-agent-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2505.12628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12628"}},"official":{"repos":["extess0/darl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visionreasoner-unified-visual-perception-and","slug":"visionreasoner-unified-visual-perception-and","title":"VisionReasoner: Unified Visual Perception and Reasoning via Reinforcement Learning","date":"2025-05-17","arxiv_id":"2505.12081","repositories_listed":3,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":13,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/visionreasoner-unified-visual-perception-and#ran","syntology_url":"https://syntology.ai/paper/2505.12081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12081"}},"official":{"repos":["dvlab-research/VisionReasoner","hiyouga/easyr1","dvlab-research/Seg-Zero"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/2505-11289","slug":"2505-11289","title":"Meta-World+: An Improved, Standardized, RL Benchmark","date":"2025-05-16","arxiv_id":"2505.11289","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2505-11289#ran","syntology_url":"https://syntology.ai/paper/2505.11289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.11289"}},"official":{"repos":["farama-foundation/metaworld"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2505-11409","slug":"2505-11409","title":"Visual Planning: Let's Think Only with Images","date":"2025-05-16","arxiv_id":"2505.11409","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/2505-11409#ran","syntology_url":"https://syntology.ai/paper/2505.11409","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.11409"}},"official":{"repos":["yix8/visualplanning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-finetunes-small","slug":"reinforcement-learning-finetunes-small","title":"Reinforcement Learning Finetunes Small Subnetworks in Large Language Models","date":"2025-05-16","arxiv_id":"2505.11711","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-finetunes-small#ran","syntology_url":"https://syntology.ai/paper/2505.11711","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.11711"}},"official":null}},{"url":"/paper/dancegrpo-unleashing-grpo-on-visual","slug":"dancegrpo-unleashing-grpo-on-visual","title":"DanceGRPO: Unleashing GRPO on Visual Generation","date":"2025-05-12","arxiv_id":"2505.07818","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dancegrpo-unleashing-grpo-on-visual#ran","syntology_url":"https://syntology.ai/paper/2505.07818","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07818"}},"official":null}},{"url":"/paper/skywork-r1v2-multimodal-hybrid-reinforcement","slug":"skywork-r1v2-multimodal-hybrid-reinforcement","title":"Skywork R1V2: Multimodal Hybrid Reinforcement Learning for Reasoning","date":"2025-04-23","arxiv_id":"2504.16656","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/skywork-r1v2-multimodal-hybrid-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2504.16656","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.16656"}},"official":{"repos":["SkyworkAI/Skywork-R1V"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ttrl-test-time-reinforcement-learning","slug":"ttrl-test-time-reinforcement-learning","title":"TTRL: Test-Time Reinforcement Learning","date":"2025-04-22","arxiv_id":"2504.16084","repositories_listed":3,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":11,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":6,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ttrl-test-time-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2504.16084","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.16084"}},"official":{"repos":["prime-rl/ttrl","tsinghuac3i/awesome-rl-reasoning-recipes"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/compile-scene-graphs-with-reinforcement","slug":"compile-scene-graphs-with-reinforcement","title":"Compile Scene Graphs with Reinforcement Learning","date":"2025-04-18","arxiv_id":"2504.13617","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/compile-scene-graphs-with-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2504.13617","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13617"}},"official":{"repos":["gpt4vision/r1-sgg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vipo-value-function-inconsistency-penalized","slug":"vipo-value-function-inconsistency-penalized","title":"VIPO: Value Function Inconsistency Penalized Offline Reinforcement Learning","date":"2025-04-16","arxiv_id":"2504.11944","repositories_listed":0,"syntology":{"n":14,"n_ran":10,"n_constructed":9,"n_ran_checked":9,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":14,"phrase":"10 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vipo-value-function-inconsistency-penalized#ran","syntology_url":"https://syntology.ai/paper/2504.11944","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11944"}},"official":null}},{"url":"/paper/perception-r1-pioneering-perception-policy","slug":"perception-r1-pioneering-perception-policy","title":"Perception-R1: Pioneering Perception Policy with Reinforcement Learning","date":"2025-04-10","arxiv_id":"2504.07954","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/perception-r1-pioneering-perception-policy#ran","syntology_url":"https://syntology.ai/paper/2504.07954","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.07954"}},"official":{"repos":["linkangheng/pr1"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/concise-reasoning-via-reinforcement-learning","slug":"concise-reasoning-via-reinforcement-learning","title":"Concise Reasoning via Reinforcement Learning","date":"2025-04-07","arxiv_id":"2504.05185","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":10,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/concise-reasoning-via-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2504.05185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.05185"}},"official":{"repos":["ai-wand/concise-reasoning"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/handling-delay-in-real-time-reinforcement","slug":"handling-delay-in-real-time-reinforcement","title":"Handling Delay in Real-Time Reinforcement Learning","date":"2025-03-30","arxiv_id":"2503.23478","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":2,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":7,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/handling-delay-in-real-time-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2503.23478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.23478"}},"official":{"repos":["avecplezir/realtime-agent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/research-learning-to-reason-with-search-for","slug":"research-learning-to-reason-with-search-for","title":"ReSearch: Learning to Reason with Search for LLMs via Reinforcement Learning","date":"2025-03-25","arxiv_id":"2503.19470","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/research-learning-to-reason-with-search-for#ran","syntology_url":"https://syntology.ai/paper/2503.19470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.19470"}},"official":null}},{"url":"/paper/cls-rl-image-classification-with-rule-based","slug":"cls-rl-image-classification-with-rule-based","title":"Think or Not Think: A Study of Explicit Thinking in Rule-Based Visual Reinforcement Fine-Tuning","date":"2025-03-20","arxiv_id":"2503.16188","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cls-rl-image-classification-with-rule-based#ran","syntology_url":"https://syntology.ai/paper/2503.16188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16188"}},"official":{"repos":["minglllli/CLS-RL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/socialjax-an-evaluation-suite-for-multi-agent","slug":"socialjax-an-evaluation-suite-for-multi-agent","title":"SocialJax: An Evaluation Suite for Multi-agent Reinforcement Learning in Sequential Social Dilemmas","date":"2025-03-18","arxiv_id":"2503.14576","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/socialjax-an-evaluation-suite-for-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2503.14576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.14576"}},"official":{"repos":["cooperativex/socialjax"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/from-demonstrations-to-rewards-alignment","slug":"from-demonstrations-to-rewards-alignment","title":"From Demonstrations to Rewards: Alignment Without Explicit Human Preferences","date":"2025-03-15","arxiv_id":"2503.13538","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/from-demonstrations-to-rewards-alignment#ran","syntology_url":"https://syntology.ai/paper/2503.13538","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.13538"}},"official":{"repos":["Hong-Lab-UMN-ECE/IRLAlignment"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/regulatory-dna-sequence-design-with","slug":"regulatory-dna-sequence-design-with","title":"Regulatory DNA sequence Design with Reinforcement Learning","date":"2025-03-11","arxiv_id":"2503.07981","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regulatory-dna-sequence-design-with#ran","syntology_url":"https://syntology.ai/paper/2503.07981","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07981"}},"official":{"repos":["yangzhao1230/taco"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/automated-proof-of-polynomial-inequalities","slug":"automated-proof-of-polynomial-inequalities","title":"Automated Proof of Polynomial Inequalities via Reinforcement Learning","date":"2025-03-09","arxiv_id":"2503.06592","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/automated-proof-of-polynomial-inequalities#ran","syntology_url":"https://syntology.ai/paper/2503.06592","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06592"}},"official":{"repos":["blliu6/appirl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/r1-zero-s-aha-moment-in-visual-reasoning-on-a","slug":"r1-zero-s-aha-moment-in-visual-reasoning-on-a","title":"R1-Zero's \"Aha Moment\" in Visual Reasoning on a 2B Non-SFT Model","date":"2025-03-07","arxiv_id":"2503.05132","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/r1-zero-s-aha-moment-in-visual-reasoning-on-a#ran","syntology_url":"https://syntology.ai/paper/2503.05132","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.05132"}},"official":{"repos":["turningpoint-ai/visualthinker-r1-zero"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/adversarial-policy-optimization-for-offline","slug":"adversarial-policy-optimization-for-offline","title":"Adversarial Policy Optimization for Offline Preference-based Reinforcement Learning","date":"2025-03-07","arxiv_id":"2503.05306","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adversarial-policy-optimization-for-offline#ran","syntology_url":"https://syntology.ai/paper/2503.05306","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.05306"}},"official":{"repos":["oh-lab/APPO"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/reinforcement-learning-with-combinatorial-1","slug":"reinforcement-learning-with-combinatorial-1","title":"Reinforcement learning with combinatorial actions for coupled restless bandits","date":"2025-03-01","arxiv_id":"2503.01919","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-combinatorial-1#ran","syntology_url":"https://syntology.ai/paper/2503.01919","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.01919"}},"official":{"repos":["lily-x/combinatorial-rmab"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deepretrieval-powerful-query-generation-for","slug":"deepretrieval-powerful-query-generation-for","title":"DeepRetrieval: Hacking Real Search Engines and Retrievers with Large Language Models via Reinforcement Learning","date":"2025-02-28","arxiv_id":"2503.00223","repositories_listed":1,"syntology":{"n":25,"n_ran":24,"n_constructed":0,"n_ran_checked":23,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":23,"n_pointer_only":0,"phrase":"24 ran (of which 0 constructed an object rather than computing a result; 23 with no instrument failure: 0 honoured, 0 violated, 23 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deepretrieval-powerful-query-generation-for#ran","syntology_url":"https://syntology.ai/paper/2503.00223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00223"}},"official":{"repos":["pat-jj/deepretrieval"],"state":"official (archive's flag): 24 ran","n_ran":24,"n_constructed":0,"n_ran_no_instrument_failure":23,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/playing-pokemon-red-via-deep-reinforcement","slug":"playing-pokemon-red-via-deep-reinforcement","title":"Playing Pokémon Red via Deep Reinforcement Learning","date":"2025-02-27","arxiv_id":"2502.19920","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/playing-pokemon-red-via-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2502.19920","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.19920"}},"official":{"repos":["MarcoMeter/neroRL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/notagen-advancing-musicality-in-symbolic","slug":"notagen-advancing-musicality-in-symbolic","title":"NotaGen: Advancing Musicality in Symbolic Music Generation with Large Language Model Training Paradigms","date":"2025-02-25","arxiv_id":"2502.18008","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/notagen-advancing-musicality-in-symbolic#ran","syntology_url":"https://syntology.ai/paper/2502.18008","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.18008"}},"official":null}},{"url":"/paper/hyperspherical-normalization-for-scalable","slug":"hyperspherical-normalization-for-scalable","title":"Hyperspherical Normalization for Scalable Deep Reinforcement Learning","date":"2025-02-21","arxiv_id":"2502.15280","repositories_listed":0,"syntology":{"n":6,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hyperspherical-normalization-for-scalable#ran","syntology_url":"https://syntology.ai/paper/2502.15280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.15280"}},"official":null}},{"url":"/paper/logic-rl-unleashing-llm-reasoning-with-rule","slug":"logic-rl-unleashing-llm-reasoning-with-rule","title":"Logic-RL: Unleashing LLM Reasoning with Rule-Based Reinforcement Learning","date":"2025-02-20","arxiv_id":"2502.14768","repositories_listed":4,"syntology":{"n":23,"n_ran":15,"n_constructed":0,"n_ran_checked":13,"n_instrument":2,"n_unverified":8,"n_honours":1,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 0 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/logic-rl-unleashing-llm-reasoning-with-rule#ran","syntology_url":"https://syntology.ai/paper/2502.14768","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14768"}},"official":{"repos":["Unakar/Logic-RL"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":6,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/maximum-entropy-reinforcement-learning-with-1","slug":"maximum-entropy-reinforcement-learning-with-1","title":"Maximum Entropy Reinforcement Learning with Diffusion Policy","date":"2025-02-17","arxiv_id":"2502.11612","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":6,"n_ran_checked":8,"n_instrument":2,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":6,"n_pointer_only":13,"phrase":"10 ran (of which 6 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/maximum-entropy-reinforcement-learning-with-1#ran","syntology_url":"https://syntology.ai/paper/2502.11612","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11612"}},"official":{"repos":["diffusionyes/maxentdp"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":6,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/active-advantage-aligned-online-reinforcement","slug":"active-advantage-aligned-online-reinforcement","title":"Active Advantage-Aligned Online Reinforcement Learning with Offline Data","date":"2025-02-11","arxiv_id":"2502.07937","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/active-advantage-aligned-online-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2502.07937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.07937"}},"official":{"repos":["xuefeng-cs/a3rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mol-moe-training-preference-guided-routers","slug":"mol-moe-training-preference-guided-routers","title":"Mol-MoE: Training Preference-Guided Routers for Molecule Generation","date":"2025-02-08","arxiv_id":"2502.05633","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mol-moe-training-preference-guided-routers#ran","syntology_url":"https://syntology.ai/paper/2502.05633","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.05633"}},"official":{"repos":["ddidacus/mol-moe"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-reinforcement-learning-with-focal","slug":"multi-agent-reinforcement-learning-with-focal","title":"Multi-Agent Reinforcement Learning with Focal Diversity Optimization","date":"2025-02-06","arxiv_id":"2502.04492","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/multi-agent-reinforcement-learning-with-focal#ran","syntology_url":"https://syntology.ai/paper/2502.04492","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.04492"}},"official":{"repos":["sftekin/rl-focal"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wolfpack-adversarial-attack-for-robust-multi","slug":"wolfpack-adversarial-attack-for-robust-multi","title":"Wolfpack Adversarial Attack for Robust Multi-Agent Reinforcement Learning","date":"2025-02-05","arxiv_id":"2502.02844","repositories_listed":1,"syntology":{"n":7,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/wolfpack-adversarial-attack-for-robust-multi#ran","syntology_url":"https://syntology.ai/paper/2502.02844","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.02844"}},"official":{"repos":["sunwoolee0504/wall"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/vintix-action-model-via-in-context","slug":"vintix-action-model-via-in-context","title":"Vintix: Action Model via In-Context Reinforcement Learning","date":"2025-01-31","arxiv_id":"2501.19400","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vintix-action-model-via-in-context#ran","syntology_url":"https://syntology.ai/paper/2501.19400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.19400"}},"official":{"repos":["dunnolab/vintix"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/on-rollouts-in-model-based-reinforcement","slug":"on-rollouts-in-model-based-reinforcement","title":"On Rollouts in Model-Based Reinforcement Learning","date":"2025-01-28","arxiv_id":"2501.16918","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-rollouts-in-model-based-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2501.16918","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.16918"}},"official":{"repos":["data-science-in-mechanical-engineering/infoprop"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-quantum-reinforcement-learning","slug":"benchmarking-quantum-reinforcement-learning","title":"Benchmarking Quantum Reinforcement Learning","date":"2025-01-27","arxiv_id":"2501.15893","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-quantum-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2501.15893","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.15893"}},"official":{"repos":["nicomeyer96/qrl-benchmark"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-general-purpose-model-free","slug":"towards-general-purpose-model-free","title":"Towards General-Purpose Model-Free Reinforcement Learning","date":"2025-01-27","arxiv_id":"2501.16142","repositories_listed":0,"syntology":{"n":9,"n_ran":5,"n_constructed":4,"n_ran_checked":4,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/towards-general-purpose-model-free#ran","syntology_url":"https://syntology.ai/paper/2501.16142","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.16142"}},"official":null}},{"url":"/paper/improving-retrieval-augmented-generation","slug":"improving-retrieval-augmented-generation","title":"Improving Retrieval-Augmented Generation through Multi-Agent Reinforcement Learning","date":"2025-01-25","arxiv_id":"2501.15228","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-retrieval-augmented-generation#ran","syntology_url":"https://syntology.ai/paper/2501.15228","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.15228"}},"official":{"repos":["chenyiqun/mmoa-rag"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/kimi-k1-5-scaling-reinforcement-learning-with","slug":"kimi-k1-5-scaling-reinforcement-learning-with","title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","date":"2025-01-22","arxiv_id":"2501.12599","repositories_listed":3,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/kimi-k1-5-scaling-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2501.12599","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.12599"}},"official":null}},{"url":"/paper/curiosity-driven-reinforcement-learning-from","slug":"curiosity-driven-reinforcement-learning-from","title":"Curiosity-Driven Reinforcement Learning from Human Feedback","date":"2025-01-20","arxiv_id":"2501.11463","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/curiosity-driven-reinforcement-learning-from#ran","syntology_url":"https://syntology.ai/paper/2501.11463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.11463"}},"official":{"repos":["ernie-research/cd-rlhf"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/constraint-adaptive-policy-switching-for","slug":"constraint-adaptive-policy-switching-for","title":"Constraint-Adaptive Policy Switching for Offline Safe Reinforcement Learning","date":"2024-12-25","arxiv_id":"2412.18946","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/constraint-adaptive-policy-switching-for#ran","syntology_url":"https://syntology.ai/paper/2412.18946","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.18946"}},"official":{"repos":["yassinech/caps"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-reinforcement-learning-for-24","slug":"multi-agent-reinforcement-learning-for-24","title":"Multi Agent Reinforcement Learning for Sequential Satellite Assignment Problems","date":"2024-12-20","arxiv_id":"2412.15573","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-agent-reinforcement-learning-for-24#ran","syntology_url":"https://syntology.ai/paper/2412.15573","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.15573"}},"official":{"repos":["Rainlabuw/rl-enabled-distributed-assignment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-reinforcement-learning-for-llm-multi","slug":"offline-reinforcement-learning-for-llm-multi","title":"Offline Reinforcement Learning for LLM Multi-Step Reasoning","date":"2024-12-20","arxiv_id":"2412.16145","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/offline-reinforcement-learning-for-llm-multi#ran","syntology_url":"https://syntology.ai/paper/2412.16145","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.16145"}},"official":{"repos":["jwhj/oreo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-safe-reinforcement-learning-using","slug":"offline-safe-reinforcement-learning-using","title":"Offline Safe Reinforcement Learning Using Trajectory Classification","date":"2024-12-19","arxiv_id":"2412.15429","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/offline-safe-reinforcement-learning-using#ran","syntology_url":"https://syntology.ai/paper/2412.15429","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.15429"}},"official":{"repos":["zgong11/TraC"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/guiding-generative-protein-language-models","slug":"guiding-generative-protein-language-models","title":"Guiding Generative Protein Language Models with Reinforcement Learning","date":"2024-12-17","arxiv_id":"2412.12979","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/guiding-generative-protein-language-models#ran","syntology_url":"https://syntology.ai/paper/2412.12979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.12979"}},"official":{"repos":["ai4pdlab/dpo_plm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/latent-safety-constrained-policy-approach-for","slug":"latent-safety-constrained-policy-approach-for","title":"Latent Safety-Constrained Policy Approach for Safe Offline Reinforcement Learning","date":"2024-12-11","arxiv_id":"2412.08794","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/latent-safety-constrained-policy-approach-for#ran","syntology_url":"https://syntology.ai/paper/2412.08794","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.08794"}},"official":{"repos":["PrajwalKoirala/LSPC-Safe-Offline-RL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-method-for-evaluating-hyperparameter","slug":"a-method-for-evaluating-hyperparameter","title":"A Method for Evaluating Hyperparameter Sensitivity in Reinforcement Learning","date":"2024-12-10","arxiv_id":"2412.07165","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-method-for-evaluating-hyperparameter#ran","syntology_url":"https://syntology.ai/paper/2412.07165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.07165"}},"official":{"repos":["jadkins99/hyperparameter_sensitivity"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-policy-as-macro","slug":"reinforcement-learning-policy-as-macro","title":"Reinforcement Learning Policy as Macro Regulator Rather than Macro Placer","date":"2024-12-10","arxiv_id":"2412.07167","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reinforcement-learning-policy-as-macro#ran","syntology_url":"https://syntology.ai/paper/2412.07167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.07167"}},"official":{"repos":["lamda-bbo/macro-regulator"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/classifier-free-guidance-in-llms-safety","slug":"classifier-free-guidance-in-llms-safety","title":"Classifier-free guidance in LLMs Safety","date":"2024-12-08","arxiv_id":"2412.06846","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/classifier-free-guidance-in-llms-safety#ran","syntology_url":"https://syntology.ai/paper/2412.06846","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06846"}},"official":{"repos":["rgsmirnov/cfg_safety_llm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/continual-deep-reinforcement-learning-with","slug":"continual-deep-reinforcement-learning-with","title":"Continual Deep Reinforcement Learning with Task-Agnostic Policy Distillation","date":"2024-11-25","arxiv_id":"2411.16532","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/continual-deep-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2411.16532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16532"}},"official":{"repos":["wabbajack1/tapd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/natural-language-reinforcement-learning-1","slug":"natural-language-reinforcement-learning-1","title":"Natural Language Reinforcement Learning","date":"2024-11-21","arxiv_id":"2411.14251","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/natural-language-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2411.14251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.14251"}},"official":{"repos":["waterhorse1/natural-language-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/doubly-mild-generalization-for-offline","slug":"doubly-mild-generalization-for-offline","title":"Doubly Mild Generalization for Offline Reinforcement Learning","date":"2024-11-12","arxiv_id":"2411.07934","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/doubly-mild-generalization-for-offline#ran","syntology_url":"https://syntology.ai/paper/2411.07934","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.07934"}},"official":{"repos":["maoyixiu/dmg"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/non-adversarial-inverse-reinforcement","slug":"non-adversarial-inverse-reinforcement","title":"Non-Adversarial Inverse Reinforcement Learning via Successor Feature Matching","date":"2024-11-11","arxiv_id":"2411.07007","repositories_listed":1,"syntology":{"n":24,"n_ran":19,"n_constructed":14,"n_ran_checked":15,"n_instrument":4,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":14,"n_pointer_only":0,"phrase":"19 ran (of which 14 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 1 violated, 14 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/non-adversarial-inverse-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2411.07007","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.07007"}},"official":{"repos":["arnavkj1995/sfm"],"state":"official (archive's flag): 19 ran","n_ran":19,"n_constructed":14,"n_ran_no_instrument_failure":15,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/constrained-latent-action-policies-for-model","slug":"constrained-latent-action-policies-for-model","title":"Constrained Latent Action Policies for Model-Based Offline Reinforcement Learning","date":"2024-11-07","arxiv_id":"2411.04562","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/constrained-latent-action-policies-for-model#ran","syntology_url":"https://syntology.ai/paper/2411.04562","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.04562"}},"official":{"repos":["marvinalles/c-lap"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchical-orchestra-of-policies","slug":"hierarchical-orchestra-of-policies","title":"Hierarchical Orchestra of Policies","date":"2024-11-05","arxiv_id":"2411.03008","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hierarchical-orchestra-of-policies#ran","syntology_url":"https://syntology.ai/paper/2411.03008","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.03008"}},"official":null}},{"url":"/paper/reinforcement-learning-gradients-as-vitamin","slug":"reinforcement-learning-gradients-as-vitamin","title":"Reinforcement Learning Gradients as Vitamin for Online Finetuning Decision Transformers","date":"2024-10-31","arxiv_id":"2410.24108","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reinforcement-learning-gradients-as-vitamin#ran","syntology_url":"https://syntology.ai/paper/2410.24108","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.24108"}},"official":{"repos":["kaiyan289/rl_as_vitamin_for_online_decision_transformers"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/from-easy-to-hard-tackling-quantum-problems","slug":"from-easy-to-hard-tackling-quantum-problems","title":"Reinforcement learning with learned gadgets to tackle hard quantum problems on real hardware","date":"2024-10-31","arxiv_id":"2411.00230","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/from-easy-to-hard-tackling-quantum-problems#ran","syntology_url":"https://syntology.ai/paper/2411.00230","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00230"}},"official":{"repos":["aqasch/gadget_rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/predicting-future-actions-of-reinforcement","slug":"predicting-future-actions-of-reinforcement","title":"Predicting Future Actions of Reinforcement Learning Agents","date":"2024-10-29","arxiv_id":"2410.22459","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/predicting-future-actions-of-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2410.22459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22459"}},"official":{"repos":["stephen-chung-mh/predict_action"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/odrl-a-benchmark-for-off-dynamics","slug":"odrl-a-benchmark-for-off-dynamics","title":"ODRL: A Benchmark for Off-Dynamics Reinforcement Learning","date":"2024-10-28","arxiv_id":"2410.20750","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/odrl-a-benchmark-for-off-dynamics#ran","syntology_url":"https://syntology.ai/paper/2410.20750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20750"}},"official":{"repos":["offdynamicsrl/off-dynamics-rl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/navigating-noisy-feedback-enhancing","slug":"navigating-noisy-feedback-enhancing","title":"Navigating Noisy Feedback: Enhancing Reinforcement Learning with Error-Prone Language Models","date":"2024-10-22","arxiv_id":"2410.17389","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/navigating-noisy-feedback-enhancing#ran","syntology_url":"https://syntology.ai/paper/2410.17389","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17389"}},"official":{"repos":["sy-shi/RLAIF_ScoreDiff"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/improve-vision-language-model-chain-of","slug":"improve-vision-language-model-chain-of","title":"Improve Vision Language Model Chain-of-thought Reasoning","date":"2024-10-21","arxiv_id":"2410.16198","repositories_listed":2,"syntology":{"n":22,"n_ran":18,"n_constructed":0,"n_ran_checked":13,"n_instrument":5,"n_unverified":4,"n_honours":0,"n_violates":3,"n_no_contract":10,"n_pointer_only":22,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 3 violated, 10 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/improve-vision-language-model-chain-of#ran","syntology_url":"https://syntology.ai/paper/2410.16198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.16198"}},"official":{"repos":["riflezhang/llava-reasoner-dpo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/intersectionzoo-eco-driving-for-benchmarking","slug":"intersectionzoo-eco-driving-for-benchmarking","title":"IntersectionZoo: Eco-driving for Benchmarking Multi-Agent Contextual Reinforcement Learning","date":"2024-10-19","arxiv_id":"2410.15221","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/intersectionzoo-eco-driving-for-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2410.15221","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.15221"}},"official":{"repos":["mit-wu-lab/IntersectionZoo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/streaming-deep-reinforcement-learning-finally","slug":"streaming-deep-reinforcement-learning-finally","title":"Streaming Deep Reinforcement Learning Finally Works","date":"2024-10-18","arxiv_id":"2410.14606","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/streaming-deep-reinforcement-learning-finally#ran","syntology_url":"https://syntology.ai/paper/2410.14606","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14606"}},"official":{"repos":["mohmdelsayed/streaming-drl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bayes-adaptive-monte-carlo-tree-search-for","slug":"bayes-adaptive-monte-carlo-tree-search-for","title":"Bayes Adaptive Monte Carlo Tree Search for Offline Model-based Reinforcement Learning","date":"2024-10-15","arxiv_id":"2410.11234","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bayes-adaptive-monte-carlo-tree-search-for#ran","syntology_url":"https://syntology.ai/paper/2410.11234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11234"}},"official":{"repos":["lucascjysdl/offline-rl-kit"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/it-takes-two-to-tango-directly-optimizing-for","slug":"it-takes-two-to-tango-directly-optimizing-for","title":"It Takes Two to Tango: Directly Optimizing for Constrained Synthesizability in Generative Molecular Design","date":"2024-10-15","arxiv_id":"2410.11527","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/it-takes-two-to-tango-directly-optimizing-for#ran","syntology_url":"https://syntology.ai/paper/2410.11527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11527"}},"official":{"repos":["schwallergroup/saturn"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/stable-hadamard-memory-revitalizing-memory","slug":"stable-hadamard-memory-revitalizing-memory","title":"Stable Hadamard Memory: Revitalizing Memory-Augmented Agents for Reinforcement Learning","date":"2024-10-14","arxiv_id":"2410.10132","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stable-hadamard-memory-revitalizing-memory#ran","syntology_url":"https://syntology.ai/paper/2410.10132","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10132"}},"official":null}},{"url":"/paper/action-gaps-and-advantages-in-continuous-time","slug":"action-gaps-and-advantages-in-continuous-time","title":"Action Gaps and Advantages in Continuous-Time Distributional Reinforcement Learning","date":"2024-10-14","arxiv_id":"2410.11022","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/action-gaps-and-advantages-in-continuous-time#ran","syntology_url":"https://syntology.ai/paper/2410.11022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11022"}},"official":null}},{"url":"/paper/improving-generalization-on-the-procgen","slug":"improving-generalization-on-the-procgen","title":"Improving Generalization on the ProcGen Benchmark with Simple Architectural Changes and Scale","date":"2024-10-13","arxiv_id":"2410.10905","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-generalization-on-the-procgen#ran","syntology_url":"https://syntology.ai/paper/2410.10905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10905"}},"official":{"repos":["anndvision/vsop-3d"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/openr-an-open-source-framework-for-advanced","slug":"openr-an-open-source-framework-for-advanced","title":"OpenR: An Open Source Framework for Advanced Reasoning with Large Language Models","date":"2024-10-12","arxiv_id":"2410.09671","repositories_listed":1,"syntology":{"n":10,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/openr-an-open-source-framework-for-advanced#ran","syntology_url":"https://syntology.ai/paper/2410.09671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09671"}},"official":null}},{"url":"/paper/reinforcement-learning-for-control-of-non","slug":"reinforcement-learning-for-control-of-non","title":"Reinforcement Learning for Control of Non-Markovian Cellular Population Dynamics","date":"2024-10-11","arxiv_id":"2410.08439","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-for-control-of-non#ran","syntology_url":"https://syntology.ai/paper/2410.08439","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08439"}},"official":{"repos":["JacobHA/RL4Dosing"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/kaleidoscope-learnable-masks-for","slug":"kaleidoscope-learnable-masks-for","title":"Kaleidoscope: Learnable Masks for Heterogeneous Multi-agent Reinforcement Learning","date":"2024-10-11","arxiv_id":"2410.08540","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":2,"n_ran_checked":8,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"11 ran (of which 2 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/kaleidoscope-learnable-masks-for#ran","syntology_url":"https://syntology.ai/paper/2410.08540","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08540"}},"official":{"repos":["lxxxxr/kaleidoscope"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":2,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"url":"/paper/coevolving-with-the-other-you-fine-tuning-llm","slug":"coevolving-with-the-other-you-fine-tuning-llm","title":"Coevolving with the Other You: Fine-Tuning LLM with Sequential Cooperative Multi-Agent Reinforcement Learning","date":"2024-10-08","arxiv_id":"2410.06101","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/coevolving-with-the-other-you-fine-tuning-llm#ran","syntology_url":"https://syntology.ai/paper/2410.06101","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.06101"}},"official":{"repos":["Harry67Hu/CORY"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/improved-off-policy-reinforcement-learning-in","slug":"improved-off-policy-reinforcement-learning-in","title":"Improved Off-policy Reinforcement Learning in Biological Sequence Design","date":"2024-10-06","arxiv_id":"2410.04461","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improved-off-policy-reinforcement-learning-in#ran","syntology_url":"https://syntology.ai/paper/2410.04461","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04461"}},"official":{"repos":["hyeonahkimm/delta_cs"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/open-world-reinforcement-learning-over-long","slug":"open-world-reinforcement-learning-over-long","title":"Open-World Reinforcement Learning over Long Short-Term Imagination","date":"2024-10-04","arxiv_id":"2410.03618","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/open-world-reinforcement-learning-over-long#ran","syntology_url":"https://syntology.ai/paper/2410.03618","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03618"}},"official":{"repos":["qiwang067/LS-Imagine"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/maniskill3-gpu-parallelized-robotics","slug":"maniskill3-gpu-parallelized-robotics","title":"ManiSkill3: GPU Parallelized Robotics Simulation and Rendering for Generalizable Embodied AI","date":"2024-10-01","arxiv_id":"2410.00425","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/maniskill3-gpu-parallelized-robotics#ran","syntology_url":"https://syntology.ai/paper/2410.00425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00425"}},"official":{"repos":["haosulab/ManiSkill"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-semantic-clustering-in-deep","slug":"exploring-semantic-clustering-in-deep","title":"Exploring Semantic Clustering in Deep Reinforcement Learning for Video Games","date":"2024-09-25","arxiv_id":"2409.17411","repositories_listed":0,"syntology":{"n":17,"n_ran":9,"n_constructed":6,"n_ran_checked":7,"n_instrument":2,"n_unverified":8,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":0,"phrase":"9 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/exploring-semantic-clustering-in-deep#ran","syntology_url":"https://syntology.ai/paper/2409.17411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.17411"}},"official":null}},{"url":"/paper/advancing-humanoid-locomotion-mastering","slug":"advancing-humanoid-locomotion-mastering","title":"Advancing Humanoid Locomotion: Mastering Challenging Terrains with Denoising World Model Learning","date":"2024-08-26","arxiv_id":"2408.14472","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/advancing-humanoid-locomotion-mastering#ran","syntology_url":"https://syntology.ai/paper/2408.14472","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.14472"}},"official":{"repos":["roboterax/humanoid-gym"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hokoff-real-game-dataset-from-honor-of-kings-1","slug":"hokoff-real-game-dataset-from-honor-of-kings-1","title":"Hokoff: Real Game Dataset from Honor of Kings and its Offline Reinforcement Learning Benchmarks","date":"2024-08-20","arxiv_id":"2408.10556","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hokoff-real-game-dataset-from-honor-of-kings-1#ran","syntology_url":"https://syntology.ai/paper/2408.10556","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.10556"}},"official":{"repos":["tencent-ailab/hokoff"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/accelerating-goal-conditioned-rl-algorithms","slug":"accelerating-goal-conditioned-rl-algorithms","title":"Accelerating Goal-Conditioned RL Algorithms and Research","date":"2024-08-20","arxiv_id":"2408.11052","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/accelerating-goal-conditioned-rl-algorithms#ran","syntology_url":"https://syntology.ai/paper/2408.11052","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.11052"}},"official":{"repos":["michalbortkiewicz/jaxgcrl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}}],"record_sha256":"254369c3552962db956eb434648dcdb71160358874ac42b69e7fb933220e3c53","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}