{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/19","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":19,"pages_in_order":152,"rows_per_page":100,"rows":[1801,1900],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/18","next":"/task/reinforcement-learning-1/papers/20","papers":[{"url":"/paper/train-once-get-a-family-state-adaptive-1","slug":"train-once-get-a-family-state-adaptive-1","title":"Train Once, Get a Family: State-Adaptive Balances for Offline-to-Online Reinforcement Learning","date":"2023-10-27","arxiv_id":"2310.17966","repositories_listed":1,"syntology":{"n":25,"n_ran":17,"n_constructed":5,"n_ran_checked":12,"n_instrument":5,"n_unverified":8,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"17 ran (of which 5 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 5 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/train-once-get-a-family-state-adaptive-1#ran","syntology_url":"https://syntology.ai/paper/2310.17966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17966"}},"official":{"repos":["leaplabthu/famo2o"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":5,"n_ran_no_instrument_failure":12,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/counterfactual-augmented-importance-sampling-1","slug":"counterfactual-augmented-importance-sampling-1","title":"Counterfactual-Augmented Importance Sampling for Semi-Offline Policy Evaluation","date":"2023-10-26","arxiv_id":"2310.17146","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/counterfactual-augmented-importance-sampling-1#ran","syntology_url":"https://syntology.ai/paper/2310.17146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17146"}},"official":{"repos":["mld3/counterfactualannot-semiope"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/crop-conservative-reward-for-model-based","slug":"crop-conservative-reward-for-model-based","title":"CROP: Conservative Reward for Model-based Offline Policy Optimization","date":"2023-10-26","arxiv_id":"2310.17245","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-when-dynamics-invariant-data","slug":"understanding-when-dynamics-invariant-data","title":"Understanding when Dynamics-Invariant Data Augmentations Benefit Model-Free Reinforcement Learning Updates","date":"2023-10-26","arxiv_id":"2310.17786","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/understanding-when-dynamics-invariant-data#ran","syntology_url":"https://syntology.ai/paper/2310.17786","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17786"}},"official":{"repos":["badger-rl/understandingdataaugmentationforrl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hyperparameter-optimization-for-multi","slug":"hyperparameter-optimization-for-multi","title":"Hyperparameter Optimization for Multi-Objective Reinforcement Learning","date":"2023-10-25","arxiv_id":"2310.16487","repositories_listed":1,"syntology":null},{"url":"/paper/corruption-robust-offline-reinforcement-2","slug":"corruption-robust-offline-reinforcement-2","title":"Corruption-Robust Offline Reinforcement Learning with General Function Approximation","date":"2023-10-23","arxiv_id":"2310.14550","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/corruption-robust-offline-reinforcement-2#ran","syntology_url":"https://syntology.ai/paper/2310.14550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14550"}},"official":{"repos":["yangrui2015/uwmsg"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diversify-question-generation-with-retrieval","slug":"diversify-question-generation-with-retrieval","title":"Diversify Question Generation with Retrieval-Augmented Style Transfer","date":"2023-10-23","arxiv_id":"2310.14503","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/diversify-question-generation-with-retrieval#ran","syntology_url":"https://syntology.ai/paper/2310.14503","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14503"}},"official":{"repos":["gouqi666/rast"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-navigation-training-autonomous-vehicles","slug":"safe-navigation-training-autonomous-vehicles","title":"Safe Navigation: Training Autonomous Vehicles using Deep Reinforcement Learning in CARLA","date":"2023-10-23","arxiv_id":"2311.10735","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safe-navigation-training-autonomous-vehicles#ran","syntology_url":"https://syntology.ai/paper/2311.10735","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10735"}},"official":{"repos":["tejas-deo/safe-navigation-training-autonomous-vehicles-using-deep-reinforcement-learning-in-carla"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/contrastive-prefence-learning-learning-from","slug":"contrastive-prefence-learning-learning-from","title":"Contrastive Preference Learning: Learning from Human Feedback without RL","date":"2023-10-20","arxiv_id":"2310.13639","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/contrastive-prefence-learning-learning-from#ran","syntology_url":"https://syntology.ai/paper/2310.13639","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.13639"}},"official":{"repos":["jhejna/cpl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sdgym-low-code-reinforcement-learning","slug":"sdgym-low-code-reinforcement-learning","title":"SDGym: Low-Code Reinforcement Learning Environments using System Dynamics Models","date":"2023-10-19","arxiv_id":"2310.12494","repositories_listed":1,"syntology":null},{"url":"/paper/vision-language-models-are-zero-shot-reward","slug":"vision-language-models-are-zero-shot-reward","title":"Vision-Language Models are Zero-Shot Reward Models for Reinforcement Learning","date":"2023-10-19","arxiv_id":"2310.12921","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vision-language-models-are-zero-shot-reward#ran","syntology_url":"https://syntology.ai/paper/2310.12921","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12921"}},"official":{"repos":["alignmentresearch/vlmrm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/accelerated-policy-gradient-on-the-nesterov","slug":"accelerated-policy-gradient-on-the-nesterov","title":"Accelerated Policy Gradient: On the Convergence Rates of the Nesterov Momentum for Reinforcement Learning","date":"2023-10-18","arxiv_id":"2310.11897","repositories_listed":1,"syntology":null},{"url":"/paper/quality-diversity-through-human-feedback","slug":"quality-diversity-through-human-feedback","title":"Quality Diversity through Human Feedback: Towards Open-Ended Diversity-Driven Optimization","date":"2023-10-18","arxiv_id":"2310.12103","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quality-diversity-through-human-feedback#ran","syntology_url":"https://syntology.ai/paper/2310.12103","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12103"}},"official":{"repos":["ld-ing/qdhf"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/non-ergodicity-in-reinforcement-learning","slug":"non-ergodicity-in-reinforcement-learning","title":"Reinforcement learning with non-ergodic reward increments: robustness via ergodicity transformations","date":"2023-10-17","arxiv_id":"2310.11335","repositories_listed":1,"syntology":null},{"url":"/paper/building-persona-consistent-dialogue-agents","slug":"building-persona-consistent-dialogue-agents","title":"Building Persona Consistent Dialogue Agents with Offline Reinforcement Learning","date":"2023-10-16","arxiv_id":"2310.10735","repositories_listed":1,"syntology":null},{"url":"/paper/amago-scalable-in-context-reinforcement","slug":"amago-scalable-in-context-reinforcement","title":"AMAGO: Scalable In-Context Reinforcement Learning for Adaptive Agents","date":"2023-10-15","arxiv_id":"2310.09971","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/amago-scalable-in-context-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2310.09971","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09971"}},"official":{"repos":["ut-austin-rpl/amago"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reduced-policy-optimization-for-continuous","slug":"reduced-policy-optimization-for-continuous","title":"Reduced Policy Optimization for Continuous Control with Hard Constraints","date":"2023-10-14","arxiv_id":"2310.09574","repositories_listed":1,"syntology":null},{"url":"/paper/metra-scalable-unsupervised-rl-with-metric","slug":"metra-scalable-unsupervised-rl-with-metric","title":"METRA: Scalable Unsupervised RL with Metric-Aware Abstraction","date":"2023-10-13","arxiv_id":"2310.08887","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/metra-scalable-unsupervised-rl-with-metric#ran","syntology_url":"https://syntology.ai/paper/2310.08887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08887"}},"official":{"repos":["seohongpark/metra"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/impact-of-multi-armed-bandit-strategies-on","slug":"impact-of-multi-armed-bandit-strategies-on","title":"Dealing with uncertainty: balancing exploration and exploitation in deep recurrent reinforcement learning","date":"2023-10-12","arxiv_id":"2310.08331","repositories_listed":1,"syntology":null},{"url":"/paper/learning-rl-policies-for-joint-beamforming","slug":"learning-rl-policies-for-joint-beamforming","title":"Learning RL-Policies for Joint Beamforming Without Exploration: A Batch Constrained Off-Policy Approach","date":"2023-10-12","arxiv_id":"2310.08660","repositories_listed":1,"syntology":null},{"url":"/paper/offline-retraining-for-online-rl-decoupled","slug":"offline-retraining-for-online-rl-decoupled","title":"Offline Retraining for Online RL: Decoupled Policy Learning to Mitigate Exploration Bias","date":"2023-10-12","arxiv_id":"2310.08558","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/offline-retraining-for-online-rl-decoupled#ran","syntology_url":"https://syntology.ai/paper/2310.08558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08558"}},"official":{"repos":["MaxSobolMark/OOO"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/virtual-augmented-reality-for-atari","slug":"virtual-augmented-reality-for-atari","title":"Virtual Augmented Reality for Atari Reinforcement Learning","date":"2023-10-12","arxiv_id":"2310.08683","repositories_listed":1,"syntology":null},{"url":"/paper/aligning-language-models-with-human-1","slug":"aligning-language-models-with-human-1","title":"Aligning Language Models with Human Preferences via a Bayesian Approach","date":"2023-10-09","arxiv_id":"2310.05782","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/aligning-language-models-with-human-1#ran","syntology_url":"https://syntology.ai/paper/2310.05782","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05782"}},"official":{"repos":["wangjs9/aligned-dpm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/deepqtest-testing-autonomous-driving-systems","slug":"deepqtest-testing-autonomous-driving-systems","title":"DeepQTest: Testing Autonomous Driving Systems with Reinforcement Learning and Real-world Weather Data","date":"2023-10-08","arxiv_id":"2310.05170","repositories_listed":1,"syntology":null},{"url":"/paper/gear-a-gpu-centric-experience-replay-system","slug":"gear-a-gpu-centric-experience-replay-system","title":"GEAR: A GPU-Centric Experience Replay System for Large Reinforcement Learning Models","date":"2023-10-08","arxiv_id":"2310.05205","repositories_listed":1,"syntology":null},{"url":"/paper/safe-deep-policy-adaptation","slug":"safe-deep-policy-adaptation","title":"Safe Deep Policy Adaptation","date":"2023-10-08","arxiv_id":"2310.08602","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-uniform-sampling-offline-reinforcement-1","slug":"beyond-uniform-sampling-offline-reinforcement-1","title":"Beyond Uniform Sampling: Offline Reinforcement Learning with Imbalanced Datasets","date":"2023-10-06","arxiv_id":"2310.04413","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/beyond-uniform-sampling-offline-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2310.04413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04413"}},"official":{"repos":["Improbable-AI/dw-offline-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/self-supervised-neuron-segmentation-with","slug":"self-supervised-neuron-segmentation-with","title":"Self-Supervised Neuron Segmentation with Multi-Agent Reinforcement Learning","date":"2023-10-06","arxiv_id":"2310.04148","repositories_listed":1,"syntology":{"n":24,"n_ran":20,"n_constructed":4,"n_ran_checked":18,"n_instrument":2,"n_unverified":4,"n_honours":2,"n_violates":1,"n_no_contract":15,"n_pointer_only":24,"phrase":"20 ran (of which 4 constructed an object rather than computing a result; 18 with no instrument failure: 2 honoured, 1 violated, 15 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/self-supervised-neuron-segmentation-with#ran","syntology_url":"https://syntology.ai/paper/2310.04148","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04148"}},"official":{"repos":["ydchen0806/dbmim"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":4,"n_ran_no_instrument_failure":18,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/lesson-learning-to-integrate-exploration","slug":"lesson-learning-to-integrate-exploration","title":"LESSON: Learning to Integrate Exploration Strategies for Reinforcement Learning via an Option Framework","date":"2023-10-05","arxiv_id":"2310.03342","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lesson-learning-to-integrate-exploration#ran","syntology_url":"https://syntology.ai/paper/2310.03342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03342"}},"official":{"repos":["beanie00/lesson"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/discovering-general-reinforcement-learning-1","slug":"discovering-general-reinforcement-learning-1","title":"Discovering General Reinforcement Learning Algorithms with Adversarial Environment Design","date":"2023-10-04","arxiv_id":"2310.02782","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/discovering-general-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2310.02782","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02782"}},"official":{"repos":["EmptyJackson/groove"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-reinforcement-learning-for-power","slug":"multi-agent-reinforcement-learning-for-power","title":"Multi-Agent Reinforcement Learning for Power Grid Topology Optimization","date":"2023-10-04","arxiv_id":"2310.02605","repositories_listed":1,"syntology":null},{"url":"/paper/learning-and-reusing-primitive-behaviours-to","slug":"learning-and-reusing-primitive-behaviours-to","title":"Learning and reusing primitive behaviours to improve Hindsight Experience Replay sample efficiency","date":"2023-10-03","arxiv_id":"2310.01827","repositories_listed":1,"syntology":null},{"url":"/paper/pgdqn-preference-guided-deep-q-network","slug":"pgdqn-preference-guided-deep-q-network","title":"PGDQN: Preference-Guided Deep Q-Network","date":"2023-10-03","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/prioritized-soft-q-decomposition-for","slug":"prioritized-soft-q-decomposition-for","title":"Prioritized Soft Q-Decomposition for Lexicographic Reinforcement Learning","date":"2023-10-03","arxiv_id":"2310.02360","repositories_listed":1,"syntology":null},{"url":"/paper/improving-dialogue-management-quality","slug":"improving-dialogue-management-quality","title":"Improving Dialogue Management: Quality Datasets vs Models","date":"2023-10-02","arxiv_id":"2310.01339","repositories_listed":1,"syntology":null},{"url":"/paper/comsd-balancing-behavioral-quality-and","slug":"comsd-balancing-behavioral-quality-and","title":"ComSD: Balancing Behavioral Quality and Diversity in Unsupervised Skill Discovery","date":"2023-09-29","arxiv_id":"2309.17203","repositories_listed":1,"syntology":null},{"url":"/paper/consistency-models-as-a-rich-and-efficient","slug":"consistency-models-as-a-rich-and-efficient","title":"Consistency Models as a Rich and Efficient Policy Class for Reinforcement Learning","date":"2023-09-29","arxiv_id":"2309.16984","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":1,"n_ran_checked":6,"n_instrument":3,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"9 ran (of which 1 constructed an object rather than computing a result; 6 with no instrument failure: 3 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/consistency-models-as-a-rich-and-efficient#ran","syntology_url":"https://syntology.ai/paper/2309.16984","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16984"}},"official":{"repos":["quantumiracle/consistency_model_for_reinforcement_learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["community","official","unlocated"]}}},{"url":"/paper/motif-intrinsic-motivation-from-artificial","slug":"motif-intrinsic-motivation-from-artificial","title":"Motif: Intrinsic Motivation from Artificial Intelligence Feedback","date":"2023-09-29","arxiv_id":"2310.00166","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/motif-intrinsic-motivation-from-artificial#ran","syntology_url":"https://syntology.ai/paper/2310.00166","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00166"}},"official":{"repos":["facebookresearch/motif"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/rladapter-bridging-large-language-models-to","slug":"rladapter-bridging-large-language-models-to","title":"AdaRefiner: Refining Decisions of Language Models with Adaptive Feedback","date":"2023-09-29","arxiv_id":"2309.17176","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rladapter-bridging-large-language-models-to#ran","syntology_url":"https://syntology.ai/paper/2309.17176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17176"}},"official":{"repos":["pku-rl/adarefiner"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-robust-offline-to-online","slug":"towards-robust-offline-to-online","title":"Towards Robust Offline-to-Online Reinforcement Learning via Uncertainty and Smoothness","date":"2023-09-29","arxiv_id":"2309.16973","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-robust-offline-to-online#ran","syntology_url":"https://syntology.ai/paper/2309.16973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16973"}},"official":{"repos":["battlewen/ro2o"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/plotmap-automated-layout-design-for-building","slug":"plotmap-automated-layout-design-for-building","title":"PlotMap: Automated Layout Design for Building Game Worlds","date":"2023-09-26","arxiv_id":"2309.15242","repositories_listed":1,"syntology":null},{"url":"/paper/recurrent-hypernetworks-are-surprisingly","slug":"recurrent-hypernetworks-are-surprisingly","title":"Recurrent Hypernetworks are Surprisingly Strong in Meta-RL","date":"2023-09-26","arxiv_id":"2309.14970","repositories_listed":1,"syntology":null},{"url":"/paper/tempo-adaptation-in-non-stationary-1","slug":"tempo-adaptation-in-non-stationary-1","title":"Tempo Adaptation in Non-stationary Reinforcement Learning","date":"2023-09-26","arxiv_id":"2309.14989","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/tempo-adaptation-in-non-stationary-1#ran","syntology_url":"https://syntology.ai/paper/2309.14989","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.14989"}},"official":{"repos":["hyunin-lee/TempoRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/guided-cooperation-in-hierarchical","slug":"guided-cooperation-in-hierarchical","title":"Guided Cooperation in Hierarchical Reinforcement Learning via Model-based Rollout","date":"2023-09-24","arxiv_id":"2309.13508","repositories_listed":1,"syntology":null},{"url":"/paper/kuaisim-a-comprehensive-simulator-for","slug":"kuaisim-a-comprehensive-simulator-for","title":"KuaiSim: A Comprehensive Simulator for Recommender Systems","date":"2023-09-22","arxiv_id":"2309.12645","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/kuaisim-a-comprehensive-simulator-for#ran","syntology_url":"https://syntology.ai/paper/2309.12645","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.12645"}},"official":{"repos":["applied-machine-learning-lab/kuaisim"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/text2reward-automated-dense-reward-function","slug":"text2reward-automated-dense-reward-function","title":"Text2Reward: Reward Shaping with Language Models for Reinforcement Learning","date":"2023-09-20","arxiv_id":"2309.11489","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/text2reward-automated-dense-reward-function#ran","syntology_url":"https://syntology.ai/paper/2309.11489","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.11489"}},"official":{"repos":["xlang-ai/text2reward"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/reward-engineering-for-generating-semi","slug":"reward-engineering-for-generating-semi","title":"Reward Engineering for Generating Semi-structured Explanation","date":"2023-09-15","arxiv_id":"2309.08347","repositories_listed":1,"syntology":null},{"url":"/paper/physics-constrained-robust-learning-of-open","slug":"physics-constrained-robust-learning-of-open","title":"Physics-constrained robust learning of open-form partial differential equations from limited and noisy data","date":"2023-09-14","arxiv_id":"2309.07672","repositories_listed":1,"syntology":null},{"url":"/paper/vapor-holonomic-legged-robot-navigation-in","slug":"vapor-holonomic-legged-robot-navigation-in","title":"VAPOR: Legged Robot Navigation in Outdoor Vegetation Using Offline Reinforcement Learning","date":"2023-09-14","arxiv_id":"2309.07832","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-reinforcement-learning-for-jumping","slug":"efficient-reinforcement-learning-for-jumping","title":"Efficient Reinforcement Learning for Jumping Monopods","date":"2023-09-13","arxiv_id":"2309.07038","repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-with-latent-diffusion-in-offline","slug":"reasoning-with-latent-diffusion-in-offline","title":"Reasoning with Latent Diffusion in Offline Reinforcement Learning","date":"2023-09-12","arxiv_id":"2309.06599","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/reasoning-with-latent-diffusion-in-offline#ran","syntology_url":"https://syntology.ai/paper/2309.06599","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.06599"}},"official":{"repos":["ldcq/ldcq"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/toward-discretization-consistent-closure","slug":"toward-discretization-consistent-closure","title":"Toward Discretization-Consistent Closure Schemes for Large Eddy Simulation Using Reinforcement Learning","date":"2023-09-12","arxiv_id":"2309.06260","repositories_listed":1,"syntology":null},{"url":"/paper/compositional-learning-of-visually-grounded","slug":"compositional-learning-of-visually-grounded","title":"Compositional Learning of Visually-Grounded Concepts Using Reinforcement","date":"2023-09-08","arxiv_id":"2309.04504","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-from-hierarchical","slug":"deep-reinforcement-learning-from-hierarchical","title":"Deep Reinforcement Learning from Hierarchical Preference Design","date":"2023-09-06","arxiv_id":"2309.02632","repositories_listed":1,"syntology":null},{"url":"/paper/natural-and-robust-walking-using","slug":"natural-and-robust-walking-using","title":"Natural and Robust Walking using Reinforcement Learning without Demonstrations in High-Dimensional Musculoskeletal Models","date":"2023-09-06","arxiv_id":"2309.02976","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-offline-policy-optimization-with","slug":"model-based-offline-policy-optimization-with","title":"Model-based Offline Policy Optimization with Adversarial Network","date":"2023-09-05","arxiv_id":"2309.02157","repositories_listed":1,"syntology":null},{"url":"/paper/parameter-and-computation-efficient-transfer-1","slug":"parameter-and-computation-efficient-transfer-1","title":"Parameter and Computation Efficient Transfer Learning for Vision-Language Pre-trained Models","date":"2023-09-04","arxiv_id":"2309.01479","repositories_listed":1,"syntology":null},{"url":"/paper/autonomous-soft-tissue-retraction-using","slug":"autonomous-soft-tissue-retraction-using","title":"Autonomous Soft Tissue Retraction Using Demonstration-Guided Reinforcement Learning","date":"2023-09-02","arxiv_id":"2309.00837","repositories_listed":1,"syntology":null},{"url":"/paper/iterative-reward-shaping-using-human-feedback","slug":"iterative-reward-shaping-using-human-feedback","title":"Iterative Reward Shaping using Human Feedback for Correcting Reward Misspecification","date":"2023-08-30","arxiv_id":"2308.15969","repositories_listed":1,"syntology":null},{"url":"/paper/improving-reinforcement-learning-training","slug":"improving-reinforcement-learning-training","title":"Improving Generalization in Reinforcement Learning Training Regimes for Social Robot Navigation","date":"2023-08-29","arxiv_id":"2308.14947","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-reinforcement-learning-training#ran","syntology_url":"https://syntology.ai/paper/2308.14947","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.14947"}},"official":{"repos":["raise-lab/soc-nav-training"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-powered-sim-to-real-transfer-for-traffic","slug":"llm-powered-sim-to-real-transfer-for-traffic","title":"Prompt to Transfer: Sim-to-Real Transfer for Traffic Signal Control with Prompt Learning","date":"2023-08-28","arxiv_id":"2308.14284","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/llm-powered-sim-to-real-transfer-for-traffic#ran","syntology_url":"https://syntology.ai/paper/2308.14284","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.14284"}},"official":{"repos":["darl-libsignal/promptgat"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/molopt-autonomous-molecular-geometry","slug":"molopt-autonomous-molecular-geometry","title":"MolOpt: Autonomous Molecular Geometry Optimization using Multi-Agent Reinforcement Learning","date":"2023-08-24","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/language-reward-modulation-for-pretraining","slug":"language-reward-modulation-for-pretraining","title":"Language Reward Modulation for Pretraining Reinforcement Learning","date":"2023-08-23","arxiv_id":"2308.12270","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-reward-modulation-for-pretraining#ran","syntology_url":"https://syntology.ai/paper/2308.12270","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12270"}},"official":{"repos":["ademiadeniji/lamp"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ramseyrl-a-framework-for-intelligent-ramsey","slug":"ramseyrl-a-framework-for-intelligent-ramsey","title":"RamseyRL: A Framework for Intelligent Ramsey Number Counterexample Searching","date":"2023-08-23","arxiv_id":"2308.11943","repositories_listed":1,"syntology":null},{"url":"/paper/lagr-seq-language-guided-reinforcement","slug":"lagr-seq-language-guided-reinforcement","title":"LaGR-SEQ: Language-Guided Reinforcement Learning with Sample-Efficient Querying","date":"2023-08-21","arxiv_id":"2308.13542","repositories_listed":1,"syntology":null},{"url":"/paper/stabilizing-unsupervised-environment-design","slug":"stabilizing-unsupervised-environment-design","title":"Stabilizing Unsupervised Environment Design with a Learned Adversary","date":"2023-08-21","arxiv_id":"2308.10797","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforcement-learning-approach-for-8","slug":"a-reinforcement-learning-approach-for-8","title":"A Reinforcement Learning Approach for Performance-aware Reduction in Power Consumption of Data Center Compute Nodes","date":"2023-08-15","arxiv_id":"2308.08069","repositories_listed":1,"syntology":null},{"url":"/paper/planning-to-learn-a-novel-algorithm-for","slug":"planning-to-learn-a-novel-algorithm-for","title":"Planning to Learn: A Novel Algorithm for Active Learning during Model-Based Planning","date":"2023-08-15","arxiv_id":"2308.08029","repositories_listed":1,"syntology":null},{"url":"/paper/acre-actor-critic-with-reward-preserving","slug":"acre-actor-critic-with-reward-preserving","title":"ACRE: Actor-Critic with Reward-Preserving Exploration","date":"2023-08-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/dialogue-for-prompting-a-policy-gradient","slug":"dialogue-for-prompting-a-policy-gradient","title":"Dialogue for Prompting: a Policy-Gradient-Based Discrete Prompt Generation for Few-shot Learning","date":"2023-08-14","arxiv_id":"2308.07272","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dialogue-for-prompting-a-policy-gradient#ran","syntology_url":"https://syntology.ai/paper/2308.07272","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.07272"}},"official":{"repos":["czx-li/DP2O"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/qdax-a-library-for-quality-diversity-and","slug":"qdax-a-library-for-quality-diversity-and","title":"QDax: A Library for Quality-Diversity and Population-based Algorithms with Hardware Acceleration","date":"2023-08-07","arxiv_id":"2308.03665","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-financial-index","slug":"reinforcement-learning-for-financial-index","title":"Reinforcement Learning for Financial Index Tracking","date":"2023-08-05","arxiv_id":"2308.02820","repositories_listed":1,"syntology":null},{"url":"/paper/bierl-a-meta-evolutionary-reinforcement","slug":"bierl-a-meta-evolutionary-reinforcement","title":"BiERL: A Meta Evolutionary Reinforcement Learning Framework via Bilevel Optimization","date":"2023-08-01","arxiv_id":"2308.01207","repositories_listed":1,"syntology":null},{"url":"/paper/qgym-a-gym-for-training-and-benchmarking-rl","slug":"qgym-a-gym-for-training-and-benchmarking-rl","title":"qgym: A Gym for Training and Benchmarking RL-Based Quantum Compilation","date":"2023-08-01","arxiv_id":"2308.02536","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-non","slug":"reinforcement-learning-based-non","title":"Reinforcement Learning-based Non-Autoregressive Solver for Traveling Salesman Problems","date":"2023-08-01","arxiv_id":"2308.00560","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-based-non#ran","syntology_url":"https://syntology.ai/paper/2308.00560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.00560"}},"official":{"repos":["xybfight/nar4tsp"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/end-to-end-reinforcement-learning-for-torque","slug":"end-to-end-reinforcement-learning-for-torque","title":"End-to-End Reinforcement Learning for Torque Based Variable Height Hopping","date":"2023-07-31","arxiv_id":"2307.16676","repositories_listed":1,"syntology":null},{"url":"/paper/drl4route-a-deep-reinforcement-learning","slug":"drl4route-a-deep-reinforcement-learning","title":"DRL4Route: A Deep Reinforcement Learning Framework for Pick-up and Delivery Route Prediction","date":"2023-07-30","arxiv_id":"2307.16246","repositories_listed":1,"syntology":null},{"url":"/paper/pimbot-policy-and-incentive-manipulation-for","slug":"pimbot-policy-and-incentive-manipulation-for","title":"PIMbot: Policy and Incentive Manipulation for Multi-Robot Reinforcement Learning in Social Dilemmas","date":"2023-07-29","arxiv_id":"2307.15944","repositories_listed":1,"syntology":null},{"url":"/paper/shrink-perturb-improves-architecture-mixing","slug":"shrink-perturb-improves-architecture-mixing","title":"Shrink-Perturb Improves Architecture Mixing during Population Based Training for Neural Architecture Search","date":"2023-07-28","arxiv_id":"2307.15621","repositories_listed":1,"syntology":null},{"url":"/paper/approximate-model-based-shielding-for-safe","slug":"approximate-model-based-shielding-for-safe","title":"Approximate Model-Based Shielding for Safe Reinforcement Learning","date":"2023-07-27","arxiv_id":"2308.00707","repositories_listed":1,"syntology":null},{"url":"/paper/mode-constrained-model-based-reinforcement","slug":"mode-constrained-model-based-reinforcement","title":"Mode-constrained Model-based Reinforcement Learning via Gaussian Processes","date":"2023-07-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-adaptation-and","slug":"reinforcement-learning-based-adaptation-and","title":"Reinforcement Learning -based Adaptation and Scheduling Methods for Multi-source DASH","date":"2023-07-25","arxiv_id":"2308.11621","repositories_listed":1,"syntology":null},{"url":"/paper/submodular-reinforcement-learning","slug":"submodular-reinforcement-learning","title":"Submodular Reinforcement Learning","date":"2023-07-25","arxiv_id":"2307.13372","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/submodular-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2307.13372","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.13372"}},"official":{"repos":["manish-pra/non-additive-rl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-effectiveness-of-offline-rl-for","slug":"on-the-effectiveness-of-offline-rl-for","title":"On the Effectiveness of Offline RL for Dialogue Response Generation","date":"2023-07-23","arxiv_id":"2307.12425","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/on-the-effectiveness-of-offline-rl-for#ran","syntology_url":"https://syntology.ai/paper/2307.12425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12425"}},"official":{"repos":["asappresearch/dialogue-offline-rl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/uncertainty-aware-grounded-action","slug":"uncertainty-aware-grounded-action","title":"Uncertainty-aware Grounded Action Transformation towards Sim-to-Real Transfer for Traffic Signal Control","date":"2023-07-23","arxiv_id":"2307.12388","repositories_listed":1,"syntology":null},{"url":"/paper/hiql-offline-goal-conditioned-rl-with-latent-1","slug":"hiql-offline-goal-conditioned-rl-with-latent-1","title":"HIQL: Offline Goal-Conditioned RL with Latent States as Actions","date":"2023-07-22","arxiv_id":"2307.11949","repositories_listed":1,"syntology":{"n":17,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/hiql-offline-goal-conditioned-rl-with-latent-1#ran","syntology_url":"https://syntology.ai/paper/2307.11949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11949"}},"official":{"repos":["seohongpark/hiql"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/joingym-an-efficient-query-optimization","slug":"joingym-an-efficient-query-optimization","title":"JoinGym: An Efficient Query Optimization Environment for Reinforcement Learning","date":"2023-07-21","arxiv_id":"2307.11704","repositories_listed":1,"syntology":null},{"url":"/paper/offline-multi-agent-reinforcement-learning-1","slug":"offline-multi-agent-reinforcement-learning-1","title":"Offline Multi-Agent Reinforcement Learning with Implicit Global-to-Local Value Regularization","date":"2023-07-21","arxiv_id":"2307.11620","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/offline-multi-agent-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2307.11620","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11620"}},"official":{"repos":["zhengyinan-air/omiga"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-potential-based-rewards-for","slug":"benchmarking-potential-based-rewards-for","title":"Benchmarking Potential Based Rewards for Learning Humanoid Locomotion","date":"2023-07-19","arxiv_id":"2307.10142","repositories_listed":1,"syntology":null},{"url":"/paper/explaining-autonomous-driving-actions-with","slug":"explaining-autonomous-driving-actions-with","title":"Explaining Autonomous Driving Actions with Visual Question Answering","date":"2023-07-19","arxiv_id":"2307.10408","repositories_listed":1,"syntology":null},{"url":"/paper/pytag-challenges-and-opportunities-for","slug":"pytag-challenges-and-opportunities-for","title":"PyTAG: Challenges and Opportunities for Reinforcement Learning in Tabletop Games","date":"2023-07-19","arxiv_id":"2307.09905","repositories_listed":1,"syntology":null},{"url":"/paper/natural-actor-critic-for-robust-reinforcement","slug":"natural-actor-critic-for-robust-reinforcement","title":"Natural Actor-Critic for Robust Reinforcement Learning with Function Approximation","date":"2023-07-17","arxiv_id":"2307.08875","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":8,"n_ran_checked":9,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":14,"phrase":"11 ran (of which 8 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/natural-actor-critic-for-robust-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2307.08875","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08875"}},"official":{"repos":["tliu1997/rnac"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":8,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/pomdp-inference-and-robust-solution-via-deep","slug":"pomdp-inference-and-robust-solution-via-deep","title":"POMDP inference and robust solution via deep reinforcement learning: An application to railway optimal maintenance","date":"2023-07-16","arxiv_id":"2307.08082","repositories_listed":1,"syntology":null},{"url":"/paper/safe-dreamerv3-safe-reinforcement-learning","slug":"safe-dreamerv3-safe-reinforcement-learning","title":"SafeDreamer: Safe Reinforcement Learning with World Models","date":"2023-07-14","arxiv_id":"2307.07176","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":4,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/safe-dreamerv3-safe-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2307.07176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.07176"}},"official":{"repos":["pku-alignment/safedreamer"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/robotic-manipulation-datasets-for-offline","slug":"robotic-manipulation-datasets-for-offline","title":"Robotic Manipulation Datasets for Offline Compositional Reinforcement Learning","date":"2023-07-13","arxiv_id":"2307.07091","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robotic-manipulation-datasets-for-offline#ran","syntology_url":"https://syntology.ai/paper/2307.07091","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.07091"}},"official":{"repos":["lifelong-ml/offline-compositional-rl-datasets"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/empowering-recommender-systems-using","slug":"empowering-recommender-systems-using","title":"Empowering recommender systems using automatically generated Knowledge Graphs and Reinforcement Learning","date":"2023-07-11","arxiv_id":"2307.04996","repositories_listed":1,"syntology":null},{"url":"/paper/payload-independent-direct-cost-learning-for","slug":"payload-independent-direct-cost-learning-for","title":"Payload-Independent Direct Cost Learning for Image Steganography","date":"2023-07-11","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/probabilistic-counterexample-guidance-for","slug":"probabilistic-counterexample-guidance-for","title":"Probabilistic Counterexample Guidance for Safer Reinforcement Learning (Extended Version)","date":"2023-07-10","arxiv_id":"2307.04927","repositories_listed":1,"syntology":null},{"url":"/paper/rltf-reinforcement-learning-from-unit-test","slug":"rltf-reinforcement-learning-from-unit-test","title":"RLTF: Reinforcement Learning from Unit Test Feedback","date":"2023-07-10","arxiv_id":"2307.04349","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/rltf-reinforcement-learning-from-unit-test#ran","syntology_url":"https://syntology.ai/paper/2307.04349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.04349"}},"official":{"repos":["zyq-scut/rltf"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/active-collection-of-well-being-and-health","slug":"active-collection-of-well-being-and-health","title":"Active Collection of Well-Being and Health Data in Mobile Devices","date":"2023-07-07","arxiv_id":null,"repositories_listed":1,"syntology":null}],"record_sha256":"915d569c69f1a1b05d375fdd60680a4f2125b42616b025b41de3706285b84656","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}