{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/10","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":10,"pages_in_order":152,"rows_per_page":100,"rows":[901,1000],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/9","next":"/task/reinforcement-learning-1/papers/11","papers":[{"url":"/paper/setting-up-a-reinforcement-learning-task-with","slug":"setting-up-a-reinforcement-learning-task-with","title":"Setting up a Reinforcement Learning Task with a Real-World Robot","date":"2018-03-19","arxiv_id":"1803.07067","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-time-series","slug":"deep-reinforcement-learning-for-time-series","title":"Deep reinforcement learning for time series: playing idealized trading games","date":"2018-03-11","arxiv_id":"1803.03916","repositories_listed":2,"syntology":null},{"url":"/paper/variance-networks-when-expectation-does-not","slug":"variance-networks-when-expectation-does-not","title":"Variance Networks: When Expectation Does Not Meet Your Expectations","date":"2018-03-10","arxiv_id":"1803.03764","repositories_listed":2,"syntology":{"n":15,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/variance-networks-when-expectation-does-not#ran","syntology_url":"https://syntology.ai/paper/1803.03764","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1803.03764"}},"official":{"repos":["da-molchanov/variance-networks"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-by-playing-solving-sparse-reward","slug":"learning-by-playing-solving-sparse-reward","title":"Learning by Playing - Solving Sparse Reward Tasks from Scratch","date":"2018-02-28","arxiv_id":"1802.10567","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-by-playing-solving-sparse-reward#ran","syntology_url":"https://syntology.ai/paper/1802.10567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.10567"}},"official":{"repos":["hu-po/pySACQ"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/model-ensemble-trust-region-policy","slug":"model-ensemble-trust-region-policy","title":"Model-Ensemble Trust-Region Policy Optimization","date":"2018-02-28","arxiv_id":"1802.10592","repositories_listed":2,"syntology":{"n":6,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 6 unverified","sample_list":"/paper/model-ensemble-trust-region-policy#ran","syntology_url":"https://syntology.ai/paper/1802.10592","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.10592"}},"official":{"repos":["thanard/me-trpo"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":6,"ran_from_kinds":[]}}},{"url":"/paper/meta-reinforcement-learning-of-structured","slug":"meta-reinforcement-learning-of-structured","title":"Meta-Reinforcement Learning of Structured Exploration Strategies","date":"2018-02-20","arxiv_id":"1802.07245","repositories_listed":2,"syntology":null},{"url":"/paper/monte-carlo-q-learning-for-general-game","slug":"monte-carlo-q-learning-for-general-game","title":"Monte Carlo Q-learning for General Game Playing","date":"2018-02-16","arxiv_id":"1802.05944","repositories_listed":2,"syntology":null},{"url":"/paper/multimodal-sentiment-analysis-with-word-level","slug":"multimodal-sentiment-analysis-with-word-level","title":"Multimodal Sentiment Analysis with Word-Level Fusion and Reinforcement Learning","date":"2018-02-03","arxiv_id":"1802.00924","repositories_listed":2,"syntology":null},{"url":"/paper/using-reinforcement-learning-to-learn-how-to","slug":"using-reinforcement-learning-to-learn-how-to","title":"Using reinforcement learning to learn how to play text-based games","date":"2018-01-06","arxiv_id":"1801.01999","repositories_listed":2,"syntology":null},{"url":"/paper/multi-timescale-memory-dynamics-in-a","slug":"multi-timescale-memory-dynamics-in-a","title":"Multi-timescale memory dynamics in a reinforcement learning network with attention-gated memory","date":"2017-12-28","arxiv_id":"1712.10062","repositories_listed":2,"syntology":null},{"url":"/paper/improving-exploration-in-evolution-strategies","slug":"improving-exploration-in-evolution-strategies","title":"Improving Exploration in Evolution Strategies for Deep Reinforcement Learning via a Population of Novelty-Seeking Agents","date":"2017-12-18","arxiv_id":"1712.06560","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-exploration-in-evolution-strategies#ran","syntology_url":"https://syntology.ai/paper/1712.06560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1712.06560"}},"official":{"repos":["uber-research/deep-neuroevolution"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/ai2-thor-an-interactive-3d-environment-for","slug":"ai2-thor-an-interactive-3d-environment-for","title":"AI2-THOR: An Interactive 3D Environment for Visual AI","date":"2017-12-14","arxiv_id":"1712.05474","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ai2-thor-an-interactive-3d-environment-for#ran","syntology_url":"https://syntology.ai/paper/1712.05474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1712.05474"}},"official":{"repos":["allenai/ai2thor"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/differentiable-lower-bound-for-expected-bleu","slug":"differentiable-lower-bound-for-expected-bleu","title":"Differentiable lower bound for expected BLEU score","date":"2017-12-13","arxiv_id":"1712.04708","repositories_listed":2,"syntology":null},{"url":"/paper/minos-multimodal-indoor-simulator-for","slug":"minos-multimodal-indoor-simulator-for","title":"MINOS: Multimodal Indoor Simulator for Navigation in Complex Environments","date":"2017-12-11","arxiv_id":"1712.03931","repositories_listed":2,"syntology":null},{"url":"/paper/noisy-natural-gradient-as-variational","slug":"noisy-natural-gradient-as-variational","title":"Noisy Natural Gradient as Variational Inference","date":"2017-12-06","arxiv_id":"1712.02390","repositories_listed":2,"syntology":null},{"url":"/paper/ai-safety-gridworlds","slug":"ai-safety-gridworlds","title":"AI Safety Gridworlds","date":"2017-11-27","arxiv_id":"1711.09883","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-sepsis","slug":"deep-reinforcement-learning-for-sepsis","title":"Deep Reinforcement Learning for Sepsis Treatment","date":"2017-11-27","arxiv_id":"1711.09602","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-for-sepsis#ran","syntology_url":"https://syntology.ai/paper/1711.09602","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1711.09602"}},"official":{"repos":["darkefyre/sepsisrl"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/action-depedent-control-variates-for-policy","slug":"action-depedent-control-variates-for-policy","title":"Action-depedent Control Variates for Policy Optimization via Stein's Identity","date":"2017-10-30","arxiv_id":"1710.11198","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/action-depedent-control-variates-for-policy#ran","syntology_url":"https://syntology.ai/paper/1710.11198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.11198"}},"official":null}},{"url":"/paper/detecting-adversarial-attacks-on-neural","slug":"detecting-adversarial-attacks-on-neural","title":"Detecting Adversarial Attacks on Neural Network Policies with Visual Foresight","date":"2017-10-02","arxiv_id":"1710.00814","repositories_listed":2,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/detecting-adversarial-attacks-on-neural#ran","syntology_url":"https://syntology.ai/paper/1710.00814","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.00814"}},"official":{"repos":["yenchenlin/rl-attack-detection"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/self-supervised-deep-reinforcement-learning","slug":"self-supervised-deep-reinforcement-learning","title":"Self-supervised Deep Reinforcement Learning with Generalized Computation Graphs for Robot Navigation","date":"2017-09-29","arxiv_id":"1709.10489","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-supervised-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1709.10489","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1709.10489"}},"official":{"repos":["gkahn13/gcg"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/deep-tamer-interactive-agent-shaping-in-high","slug":"deep-tamer-interactive-agent-shaping-in-high","title":"Deep TAMER: Interactive Agent Shaping in High-Dimensional State Spaces","date":"2017-09-28","arxiv_id":"1709.10163","repositories_listed":2,"syntology":null},{"url":"/paper/towards-optimally-decentralized-multi-robot","slug":"towards-optimally-decentralized-multi-robot","title":"Towards Optimally Decentralized Multi-Robot Collision Avoidance via Deep Reinforcement Learning","date":"2017-09-28","arxiv_id":"1709.10082","repositories_listed":2,"syntology":null},{"url":"/paper/a-benchmark-environment-motivated-by","slug":"a-benchmark-environment-motivated-by","title":"A Benchmark Environment Motivated by Industrial Control Problems","date":"2017-09-27","arxiv_id":"1709.09480","repositories_listed":2,"syntology":null},{"url":"/paper/neural-optimizer-search-with-reinforcement","slug":"neural-optimizer-search-with-reinforcement","title":"Neural Optimizer Search with Reinforcement Learning","date":"2017-09-21","arxiv_id":"1709.07417","repositories_listed":2,"syntology":null},{"url":"/paper/tensorflow-agents-efficient-batched","slug":"tensorflow-agents-efficient-batched","title":"TensorFlow Agents: Efficient Batched Reinforcement Learning in TensorFlow","date":"2017-09-08","arxiv_id":"1709.02878","repositories_listed":2,"syntology":null},{"url":"/paper/mean-actor-critic","slug":"mean-actor-critic","title":"Mean Actor Critic","date":"2017-09-01","arxiv_id":"1709.00503","repositories_listed":2,"syntology":null},{"url":"/paper/deeppath-a-reinforcement-learning-method-for","slug":"deeppath-a-reinforcement-learning-method-for","title":"DeepPath: A Reinforcement Learning Method for Knowledge Graph Reasoning","date":"2017-07-20","arxiv_id":"1707.06690","repositories_listed":2,"syntology":null},{"url":"/paper/imagination-augmented-agents-for-deep","slug":"imagination-augmented-agents-for-deep","title":"Imagination-Augmented Agents for Deep Reinforcement Learning","date":"2017-07-19","arxiv_id":"1707.06203","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/imagination-augmented-agents-for-deep#ran","syntology_url":"https://syntology.ai/paper/1707.06203","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1707.06203"}},"official":null}},{"url":"/paper/value-prediction-network","slug":"value-prediction-network","title":"Value Prediction Network","date":"2017-07-11","arxiv_id":"1707.03497","repositories_listed":2,"syntology":null},{"url":"/paper/elf-an-extensive-lightweight-and-flexible","slug":"elf-an-extensive-lightweight-and-flexible","title":"ELF: An Extensive, Lightweight and Flexible Research Platform for Real-time Strategy Games","date":"2017-07-04","arxiv_id":"1707.01067","repositories_listed":2,"syntology":null},{"url":"/paper/the-atari-grand-challenge-dataset","slug":"the-atari-grand-challenge-dataset","title":"The Atari Grand Challenge Dataset","date":"2017-05-31","arxiv_id":"1705.10998","repositories_listed":2,"syntology":null},{"url":"/paper/ask-the-right-questions-active-question","slug":"ask-the-right-questions-active-question","title":"Ask the Right Questions: Active Question Reformulation with Reinforcement Learning","date":"2017-05-22","arxiv_id":"1705.07830","repositories_listed":2,"syntology":null},{"url":"/paper/task-oriented-query-reformulation-with-1","slug":"task-oriented-query-reformulation-with-1","title":"Task-Oriented Query Reformulation with Reinforcement Learning","date":"2017-04-15","arxiv_id":"1704.04572","repositories_listed":2,"syntology":null},{"url":"/paper/stochastic-neural-networks-for-hierarchical","slug":"stochastic-neural-networks-for-hierarchical","title":"Stochastic Neural Networks for Hierarchical Reinforcement Learning","date":"2017-04-10","arxiv_id":"1704.03012","repositories_listed":2,"syntology":null},{"url":"/paper/learning-visual-servoing-with-deep-features","slug":"learning-visual-servoing-with-deep-features","title":"Learning Visual Servoing with Deep Features and Fitted Q-Iteration","date":"2017-03-31","arxiv_id":"1703.11000","repositories_listed":2,"syntology":null},{"url":"/paper/dynamic-computational-time-for-visual","slug":"dynamic-computational-time-for-visual","title":"Dynamic Computational Time for Visual Attention","date":"2017-03-30","arxiv_id":"1703.10332","repositories_listed":2,"syntology":null},{"url":"/paper/socially-aware-motion-planning-with-deep","slug":"socially-aware-motion-planning-with-deep","title":"Socially Aware Motion Planning with Deep Reinforcement Learning","date":"2017-03-26","arxiv_id":"1703.08862","repositories_listed":2,"syntology":null},{"url":"/paper/active-one-shot-learning","slug":"active-one-shot-learning","title":"Active One-shot Learning","date":"2017-02-21","arxiv_id":"1702.06559","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/active-one-shot-learning#ran","syntology_url":"https://syntology.ai/paper/1702.06559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1702.06559"}},"official":{"repos":["markpwoodward/active_osl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autonomous-braking-system-via-deep","slug":"autonomous-braking-system-via-deep","title":"Autonomous Braking System via Deep Reinforcement Learning","date":"2017-02-08","arxiv_id":"1702.02302","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-an-overview","slug":"deep-reinforcement-learning-an-overview","title":"Deep Reinforcement Learning: An Overview","date":"2017-01-25","arxiv_id":"1701.07274","repositories_listed":2,"syntology":null},{"url":"/paper/learning-light-transport-the-reinforced-way","slug":"learning-light-transport-the-reinforced-way","title":"Learning Light Transport the Reinforced Way","date":"2017-01-25","arxiv_id":"1701.07403","repositories_listed":2,"syntology":null},{"url":"/paper/regularizing-neural-networks-by-penalizing","slug":"regularizing-neural-networks-by-penalizing","title":"Regularizing Neural Networks by Penalizing Confident Output Distributions","date":"2017-01-23","arxiv_id":"1701.06548","repositories_listed":2,"syntology":null},{"url":"/paper/learning-through-dialogue-interactions-by","slug":"learning-through-dialogue-interactions-by","title":"Learning through Dialogue Interactions by Asking Questions","date":"2016-12-15","arxiv_id":"1612.04936","repositories_listed":2,"syntology":null},{"url":"/paper/dialogue-learning-with-human-in-the-loop","slug":"dialogue-learning-with-human-in-the-loop","title":"Dialogue Learning With Human-In-The-Loop","date":"2016-11-29","arxiv_id":"1611.09823","repositories_listed":2,"syntology":null},{"url":"/paper/q-prop-sample-efficient-policy-gradient-with","slug":"q-prop-sample-efficient-policy-gradient-with","title":"Q-Prop: Sample-Efficient Policy Gradient with An Off-Policy Critic","date":"2016-11-07","arxiv_id":"1611.02247","repositories_listed":2,"syntology":null},{"url":"/paper/modular-multitask-reinforcement-learning-with","slug":"modular-multitask-reinforcement-learning-with","title":"Modular Multitask Reinforcement Learning with Policy Sketches","date":"2016-11-06","arxiv_id":"1611.01796","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/modular-multitask-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/1611.01796","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1611.01796"}},"official":{"repos":["jacobandreas/psketch"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/multi-objective-deep-reinforcement-learning","slug":"multi-objective-deep-reinforcement-learning","title":"Multi-Objective Deep Reinforcement Learning","date":"2016-10-09","arxiv_id":"1610.02707","repositories_listed":2,"syntology":null},{"url":"/paper/target-driven-visual-navigation-in-indoor","slug":"target-driven-visual-navigation-in-indoor","title":"Target-driven Visual Navigation in Indoor Scenes using Deep Reinforcement Learning","date":"2016-09-16","arxiv_id":"1609.05143","repositories_listed":2,"syntology":null},{"url":"/paper/a-greedy-approach-to-adapting-the-trace","slug":"a-greedy-approach-to-adapting-the-trace","title":"A Greedy Approach to Adapting the Trace Parameter for Temporal Difference Learning","date":"2016-07-02","arxiv_id":"1607.00446","repositories_listed":2,"syntology":null},{"url":"/paper/cooperative-inverse-reinforcement-learning","slug":"cooperative-inverse-reinforcement-learning","title":"Cooperative Inverse Reinforcement Learning","date":"2016-06-09","arxiv_id":"1606.03137","repositories_listed":2,"syntology":null},{"url":"/paper/vime-variational-information-maximizing","slug":"vime-variational-information-maximizing","title":"VIME: Variational Information Maximizing Exploration","date":"2016-05-31","arxiv_id":"1605.09674","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vime-variational-information-maximizing#ran","syntology_url":"https://syntology.ai/paper/1605.09674","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1605.09674"}},"official":null}},{"url":"/paper/learning-and-policy-search-in-stochastic","slug":"learning-and-policy-search-in-stochastic","title":"Learning and Policy Search in Stochastic Dynamical Systems with Bayesian Neural Networks","date":"2016-05-23","arxiv_id":"1605.07127","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-in-large-discrete","slug":"deep-reinforcement-learning-in-large-discrete","title":"Deep Reinforcement Learning in Large Discrete Action Spaces","date":"2015-12-24","arxiv_id":"1512.07679","repositories_listed":2,"syntology":null},{"url":"/paper/increasing-the-action-gap-new-operators-for","slug":"increasing-the-action-gap-new-operators-for","title":"Increasing the Action Gap: New Operators for Reinforcement Learning","date":"2015-12-15","arxiv_id":"1512.04860","repositories_listed":2,"syntology":null},{"url":"/paper/mazebase-a-sandbox-for-learning-from-games","slug":"mazebase-a-sandbox-for-learning-from-games","title":"MazeBase: A Sandbox for Learning from Games","date":"2015-11-23","arxiv_id":"1511.07401","repositories_listed":2,"syntology":null},{"url":"/paper/doubly-robust-off-policy-value-evaluation-for","slug":"doubly-robust-off-policy-value-evaluation-for","title":"Doubly Robust Off-policy Value Evaluation for Reinforcement Learning","date":"2015-11-11","arxiv_id":"1511.03722","repositories_listed":2,"syntology":null},{"url":"/paper/variational-information-maximisation-for","slug":"variational-information-maximisation-for","title":"Variational Information Maximisation for Intrinsically Motivated Reinforcement Learning","date":"2015-09-29","arxiv_id":"1509.08731","repositories_listed":2,"syntology":null},{"url":"/paper/maximum-entropy-deep-inverse-reinforcement","slug":"maximum-entropy-deep-inverse-reinforcement","title":"Maximum Entropy Deep Inverse Reinforcement Learning","date":"2015-07-17","arxiv_id":"1507.04888","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/maximum-entropy-deep-inverse-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1507.04888","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1507.04888"}},"official":null}},{"url":"/paper/a-monte-carlo-aixi-approximation","slug":"a-monte-carlo-aixi-approximation","title":"A Monte Carlo AIXI Approximation","date":"2009-09-04","arxiv_id":"0909.0801","repositories_listed":2,"syntology":null},{"url":"/paper/quantum-reinforcement-learning","slug":"quantum-reinforcement-learning","title":"Quantum reinforcement learning","date":"2008-10-21","arxiv_id":"0810.3828","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quantum-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/0810.3828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"0810.3828"}},"official":null}},{"url":"/paper/bridging-the-gap-in-vision-language-models-in","slug":"bridging-the-gap-in-vision-language-models-in","title":"Bridging the Gap in Vision Language Models in Identifying Unsafe Concepts Across Modalities","date":"2025-07-15","arxiv_id":"2507.11155","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-robustness-of-tractoracle","slug":"exploring-the-robustness-of-tractoracle","title":"Exploring the robustness of TractOracle methods in RL-based tractography","date":"2025-07-15","arxiv_id":"2507.11486","repositories_listed":1,"syntology":null},{"url":"/paper/high-throughput-distributed-reinforcement","slug":"high-throughput-distributed-reinforcement","title":"High-Throughput Distributed Reinforcement Learning via Adaptive Policy Synchronization","date":"2025-07-15","arxiv_id":"2507.10990","repositories_listed":1,"syntology":null},{"url":"/paper/personalized-exercise-recommendation-with","slug":"personalized-exercise-recommendation-with","title":"Personalized Exercise Recommendation with Semantically-Grounded Knowledge Tracing","date":"2025-07-15","arxiv_id":"2507.11060","repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-or-memorization-unreliable-results","slug":"reasoning-or-memorization-unreliable-results","title":"Reasoning or Memorization? Unreliable Results of Reinforcement Learning Due to Data Contamination","date":"2025-07-14","arxiv_id":"2507.10532","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reasoning-or-memorization-unreliable-results#ran","syntology_url":"https://syntology.ai/paper/2507.10532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.10532"}},"official":{"repos":["wumingqi/LLM-Math-Evaluation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-with-gradient-1","slug":"deep-reinforcement-learning-with-gradient-1","title":"Deep Reinforcement Learning with Gradient Eligibility Traces","date":"2025-07-12","arxiv_id":"2507.09087","repositories_listed":1,"syntology":null},{"url":"/paper/a-practical-two-stage-recipe-for-mathematical","slug":"a-practical-two-stage-recipe-for-mathematical","title":"A Practical Two-Stage Recipe for Mathematical LLMs: Maximizing Accuracy with SFT and Efficiency with Reinforcement Learning","date":"2025-07-11","arxiv_id":"2507.08267","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-practical-two-stage-recipe-for-mathematical#ran","syntology_url":"https://syntology.ai/paper/2507.08267","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.08267"}},"official":{"repos":["analokmaus/kaggle-aimo2-fast-math-r1"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-rl-to-long-videos","slug":"scaling-rl-to-long-videos","title":"Scaling RL to Long Videos","date":"2025-07-10","arxiv_id":"2507.07966","repositories_listed":1,"syntology":{"n":24,"n_ran":18,"n_constructed":1,"n_ran_checked":14,"n_instrument":4,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":13,"n_pointer_only":1,"phrase":"18 ran (of which 1 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 1 violated, 13 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/scaling-rl-to-long-videos#ran","syntology_url":"https://syntology.ai/paper/2507.07966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.07966"}},"official":{"repos":["hiyouga/easyr1"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/autotriton-automatic-triton-programming-with","slug":"autotriton-automatic-triton-programming-with","title":"AutoTriton: Automatic Triton Programming with Reinforcement Learning in LLMs","date":"2025-07-08","arxiv_id":"2507.05687","repositories_listed":1,"syntology":null},{"url":"/paper/gta1-gui-test-time-scaling-agent","slug":"gta1-gui-test-time-scaling-agent","title":"GTA1: GUI Test-time Scaling Agent","date":"2025-07-08","arxiv_id":"2507.05791","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gta1-gui-test-time-scaling-agent#ran","syntology_url":"https://syntology.ai/paper/2507.05791","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.05791"}},"official":{"repos":["yan98/gta1"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/high-resolution-visual-reasoning-via-multi","slug":"high-resolution-visual-reasoning-via-multi","title":"High-Resolution Visual Reasoning via Multi-Turn Grounding-Based Reinforcement Learning","date":"2025-07-08","arxiv_id":"2507.05920","repositories_listed":1,"syntology":null},{"url":"/paper/generalized-adaptive-transfer-network","slug":"generalized-adaptive-transfer-network","title":"Generalized Adaptive Transfer Network: Enhancing Transfer Learning in Reinforcement Learning Across Domains","date":"2025-07-02","arxiv_id":"2507.03026","repositories_listed":1,"syntology":null},{"url":"/paper/kwai-keye-vl-technical-report","slug":"kwai-keye-vl-technical-report","title":"Kwai Keye-VL Technical Report","date":"2025-07-02","arxiv_id":"2507.01949","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 3 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/kwai-keye-vl-technical-report#ran","syntology_url":"https://syntology.ai/paper/2507.01949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.01949"}},"official":{"repos":["kwai-keye/keye"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/constructing-non-markovian-decision-process","slug":"constructing-non-markovian-decision-process","title":"Constructing Non-Markovian Decision Process via History Aggregator","date":"2025-06-30","arxiv_id":"2506.24026","repositories_listed":1,"syntology":null},{"url":"/paper/rag-r1-incentivize-the-search-and-reasoning","slug":"rag-r1-incentivize-the-search-and-reasoning","title":"RAG-R1 : Incentivize the Search and Reasoning Capabilities of LLMs through Multi-query Parallelism","date":"2025-06-30","arxiv_id":"2507.02962","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rag-r1-incentivize-the-search-and-reasoning#ran","syntology_url":"https://syntology.ai/paper/2507.02962","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.02962"}},"official":null}},{"url":"/paper/homogenization-of-multi-agent-learning","slug":"homogenization-of-multi-agent-learning","title":"Homogenization of Multi-agent Learning Dynamics in Finite-state Markov Games","date":"2025-06-26","arxiv_id":"2506.21079","repositories_listed":1,"syntology":null},{"url":"/paper/humanomniv2-from-understanding-to-omni-modal","slug":"humanomniv2-from-understanding-to-omni-modal","title":"HumanOmniV2: From Understanding to Omni-Modal Reasoning with Context","date":"2025-06-26","arxiv_id":"2506.21277","repositories_listed":1,"syntology":null},{"url":"/paper/complex-model-transformations-by","slug":"complex-model-transformations-by","title":"Complex Model Transformations by Reinforcement Learning with Uncertain Human Guidance","date":"2025-06-25","arxiv_id":"2506.20883","repositories_listed":1,"syntology":null},{"url":"/paper/diffucoder-understanding-and-improving-masked","slug":"diffucoder-understanding-and-improving-masked","title":"DiffuCoder: Understanding and Improving Masked Diffusion Models for Code Generation","date":"2025-06-25","arxiv_id":"2506.20639","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/diffucoder-understanding-and-improving-masked#ran","syntology_url":"https://syntology.ai/paper/2506.20639","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.20639"}},"official":{"repos":["apple/ml-diffucoder"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/iranker-towards-ranking-foundation-model","slug":"iranker-towards-ranking-foundation-model","title":"IRanker: Towards Ranking Foundation Model","date":"2025-06-25","arxiv_id":"2506.21638","repositories_listed":1,"syntology":null},{"url":"/paper/octothinker-mid-training-incentivizes","slug":"octothinker-mid-training-incentivizes","title":"OctoThinker: Mid-training Incentivizes Reinforcement Learning Scaling","date":"2025-06-25","arxiv_id":"2506.20512","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":4,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/octothinker-mid-training-incentivizes#ran","syntology_url":"https://syntology.ai/paper/2506.20512","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.20512"}},"official":{"repos":["gair-nlp/octothinker"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-increases-wind-farm","slug":"reinforcement-learning-increases-wind-farm","title":"Reinforcement Learning Increases Wind Farm Power Production by Enabling Closed-Loop Collaborative Control","date":"2025-06-25","arxiv_id":"2506.20554","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-increases-wind-farm#ran","syntology_url":"https://syntology.ai/paper/2506.20554","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.20554"}},"official":{"repos":["admole/wind-rl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/knowrl-exploring-knowledgeable-reinforcement","slug":"knowrl-exploring-knowledgeable-reinforcement","title":"KnowRL: Exploring Knowledgeable Reinforcement Learning for Factuality","date":"2025-06-24","arxiv_id":"2506.19807","repositories_listed":1,"syntology":null},{"url":"/paper/partially-observable-residual-reinforcement","slug":"partially-observable-residual-reinforcement","title":"Partially Observable Residual Reinforcement Learning for PV-Inverter-Based Voltage Control in Distribution Grids","date":"2025-06-24","arxiv_id":"2506.19353","repositories_listed":1,"syntology":null},{"url":"/paper/confucius3-math-a-lightweight-high","slug":"confucius3-math-a-lightweight-high","title":"Confucius3-Math: A Lightweight High-Performance Reasoning LLM for Chinese K-12 Mathematics Learning","date":"2025-06-23","arxiv_id":"2506.18330","repositories_listed":1,"syntology":null},{"url":"/paper/longwriter-zero-mastering-ultra-long-text","slug":"longwriter-zero-mastering-ultra-long-text","title":"LongWriter-Zero: Mastering Ultra-Long Text Generation via Reinforcement Learning","date":"2025-06-23","arxiv_id":"2506.18841","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":6,"n_instrument":6,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/longwriter-zero-mastering-ultra-long-text#ran","syntology_url":"https://syntology.ai/paper/2506.18841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.18841"}},"official":null}},{"url":"/paper/graphs-meet-ai-agents-taxonomy-progress-and","slug":"graphs-meet-ai-agents-taxonomy-progress-and","title":"Graphs Meet AI Agents: Taxonomy, Progress, and Future Opportunities","date":"2025-06-22","arxiv_id":"2506.18019","repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-actor-critic-for-adversarial","slug":"off-policy-actor-critic-for-adversarial","title":"Off-Policy Actor-Critic for Adversarial Observation Robustness: Virtual Alternative Training via Symmetric Policy Evaluation","date":"2025-06-20","arxiv_id":"2506.16753","repositories_listed":1,"syntology":null},{"url":"/paper/sparse-reg-improving-sample-complexity-in","slug":"sparse-reg-improving-sample-complexity-in","title":"Sparse-Reg: Improving Sample Complexity in Offline Reinforcement Learning using Sparsity","date":"2025-06-20","arxiv_id":"2506.17155","repositories_listed":1,"syntology":null},{"url":"/paper/a-production-scheduling-framework-for","slug":"a-production-scheduling-framework-for","title":"A Production Scheduling Framework for Reinforcement Learning Under Real-World Constraints","date":"2025-06-16","arxiv_id":"2506.13566","repositories_listed":1,"syntology":null},{"url":"/paper/metis-rise-rl-incentivizes-and-sft-enhances","slug":"metis-rise-rl-incentivizes-and-sft-enhances","title":"Metis-RISE: RL Incentivizes and SFT Enhances Multimodal Reasoning Model Learning","date":"2025-06-16","arxiv_id":"2506.13056","repositories_listed":1,"syntology":null},{"url":"/paper/minimax-m1-scaling-test-time-compute","slug":"minimax-m1-scaling-test-time-compute","title":"MiniMax-M1: Scaling Test-Time Compute Efficiently with Lightning Attention","date":"2025-06-16","arxiv_id":"2506.13585","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/minimax-m1-scaling-test-time-compute#ran","syntology_url":"https://syntology.ai/paper/2506.13585","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.13585"}},"official":{"repos":["minimax-ai/minimax-m1"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/overcoming-overfitting-in-reinforcement","slug":"overcoming-overfitting-in-reinforcement","title":"Overcoming Overfitting in Reinforcement Learning via Gaussian Process Diffusion Policy","date":"2025-06-16","arxiv_id":"2506.13111","repositories_listed":1,"syntology":null},{"url":"/paper/timemaster-training-time-series-multimodal","slug":"timemaster-training-time-series-multimodal","title":"TimeMaster: Training Time-Series Multimodal LLMs to Reason via Reinforcement Learning","date":"2025-06-16","arxiv_id":"2506.13705","repositories_listed":1,"syntology":null},{"url":"/paper/value-free-policy-optimization-via-reward","slug":"value-free-policy-optimization-via-reward","title":"Value-Free Policy Optimization via Reward Partitioning","date":"2025-06-16","arxiv_id":"2506.13702","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-rating-based-reinforcement-learning","slug":"enhancing-rating-based-reinforcement-learning","title":"Enhancing Rating-Based Reinforcement Learning to Effectively Leverage Feedback from Large Vision-Language Models","date":"2025-06-15","arxiv_id":"2506.12822","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/enhancing-rating-based-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2506.12822","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.12822"}},"official":{"repos":["tunglm2203/erlvlm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/soundmind-rl-incentivized-logic-reasoning-for","slug":"soundmind-rl-incentivized-logic-reasoning-for","title":"SoundMind: RL-Incentivized Logic Reasoning for Audio-Language Models","date":"2025-06-15","arxiv_id":"2506.12935","repositories_listed":1,"syntology":{"n":16,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/soundmind-rl-incentivized-logic-reasoning-for#ran","syntology_url":"https://syntology.ai/paper/2506.12935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.12935"}},"official":{"repos":["xid32/soundmind"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/dr-sac-distributionally-robust-soft-actor","slug":"dr-sac-distributionally-robust-soft-actor","title":"DR-SAC: Distributionally Robust Soft Actor-Critic for Reinforcement Learning under Uncertainty","date":"2025-06-14","arxiv_id":"2506.12622","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/dr-sac-distributionally-robust-soft-actor#ran","syntology_url":"https://syntology.ai/paper/2506.12622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.12622"}},"official":{"repos":["lemutisme/dr-sac"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/treerl-llm-reinforcement-learning-with-on","slug":"treerl-llm-reinforcement-learning-with-on","title":"TreeRL: LLM Reinforcement Learning with On-Policy Tree Search","date":"2025-06-13","arxiv_id":"2506.11902","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/treerl-llm-reinforcement-learning-with-on#ran","syntology_url":"https://syntology.ai/paper/2506.11902","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.11902"}},"official":{"repos":["thudm/treerl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-pre-training-on-unlabeled-images-using","slug":"visual-pre-training-on-unlabeled-images-using","title":"Visual Pre-Training on Unlabeled Images using Reinforcement Learning","date":"2025-06-13","arxiv_id":"2506.11967","repositories_listed":1,"syntology":null}],"record_sha256":"88a8426381b723092171e9c38d01a9c2b818b331def78639cc537897eb2d2648","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}