{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/q-learning/papers/2","list_of":"/task/q-learning","task":"Q-Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":20,"rows_per_page":100,"rows":[101,200],"of":1918,"counts":{"archive_papers_tagged":1918,"with_a_code_link":463,"where_syntology_ran_a_sample":119,"not_listed_spam_title":0,"listed":1918,"listed_where_code_ran":119,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":102,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":102,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/q-learning","prev":"/task/q-learning","next":"/task/q-learning/papers/3","papers":[{"url":"/paper/monte-carlo-q-learning-for-general-game","slug":"monte-carlo-q-learning-for-general-game","title":"Monte Carlo Q-learning for General Game Playing","date":"2018-02-16","arxiv_id":"1802.05944","repositories_listed":2,"syntology":null},{"url":"/paper/using-deep-q-learning-to-understand-the-tax","slug":"using-deep-q-learning-to-understand-the-tax","title":"Using deep Q-learning to understand the tax evasion behavior of risk-averse firms","date":"2018-01-29","arxiv_id":"1801.09466","repositories_listed":2,"syntology":null},{"url":"/paper/improving-exploration-in-evolution-strategies","slug":"improving-exploration-in-evolution-strategies","title":"Improving Exploration in Evolution Strategies for Deep Reinforcement Learning via a Population of Novelty-Seeking Agents","date":"2017-12-18","arxiv_id":"1712.06560","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-exploration-in-evolution-strategies#ran","syntology_url":"https://syntology.ai/paper/1712.06560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1712.06560"}},"official":{"repos":["uber-research/deep-neuroevolution"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/self-supervised-deep-reinforcement-learning","slug":"self-supervised-deep-reinforcement-learning","title":"Self-supervised Deep Reinforcement Learning with Generalized Computation Graphs for Robot Navigation","date":"2017-09-29","arxiv_id":"1709.10489","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-supervised-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1709.10489","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1709.10489"}},"official":{"repos":["gkahn13/gcg"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/q-prop-sample-efficient-policy-gradient-with","slug":"q-prop-sample-efficient-policy-gradient-with","title":"Q-Prop: Sample-Efficient Policy Gradient with An Off-Policy Critic","date":"2016-11-07","arxiv_id":"1611.02247","repositories_listed":2,"syntology":null},{"url":"/paper/increasing-the-action-gap-new-operators-for","slug":"increasing-the-action-gap-new-operators-for","title":"Increasing the Action Gap: New Operators for Reinforcement Learning","date":"2015-12-15","arxiv_id":"1512.04860","repositories_listed":2,"syntology":null},{"url":"/paper/personalized-exercise-recommendation-with","slug":"personalized-exercise-recommendation-with","title":"Personalized Exercise Recommendation with Semantically-Grounded Knowledge Tracing","date":"2025-07-15","arxiv_id":"2507.11060","repositories_listed":1,"syntology":null},{"url":"/paper/addq-adaptive-distributional-double-q","slug":"addq-adaptive-distributional-double-q","title":"ADDQ: Adaptive Distributional Double Q-Learning","date":"2025-06-24","arxiv_id":"2506.19478","repositories_listed":1,"syntology":null},{"url":"/paper/inverse-q-learning-done-right-offline","slug":"inverse-q-learning-done-right-offline","title":"Inverse Q-Learning Done Right: Offline Imitation Learning in $Q^π$-Realizable MDPs","date":"2025-05-26","arxiv_id":"2505.19946","repositories_listed":1,"syntology":null},{"url":"/paper/distributionally-robust-deep-q-learning","slug":"distributionally-robust-deep-q-learning","title":"Distributionally Robust Deep Q-Learning","date":"2025-05-25","arxiv_id":"2505.19058","repositories_listed":1,"syntology":null},{"url":"/paper/a-critical-assessment-of-reinforcement","slug":"a-critical-assessment-of-reinforcement","title":"A critical assessment of reinforcement learning methods for microswimmer navigation in complex flows","date":"2025-05-08","arxiv_id":"2505.05525","repositories_listed":1,"syntology":null},{"url":"/paper/meta-black-box-optimization-through-offline-q","slug":"meta-black-box-optimization-through-offline-q","title":"Meta-Black-Box-Optimization through Offline Q-function Learning","date":"2025-05-04","arxiv_id":"2505.02010","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/meta-black-box-optimization-through-offline-q#ran","syntology_url":"https://syntology.ai/paper/2505.02010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02010"}},"official":{"repos":["metaevo/q-mamba"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/deep-reinforcement-learning-algorithms-for-1","slug":"deep-reinforcement-learning-algorithms-for-1","title":"Deep Reinforcement Learning Algorithms for Option Hedging","date":"2025-04-07","arxiv_id":"2504.05521","repositories_listed":1,"syntology":null},{"url":"/paper/omniecon-nexus-global-microeconomic","slug":"omniecon-nexus-global-microeconomic","title":"OmniEcon Nexus: Global Microeconomic Simulation Engine","date":"2025-04-07","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/pairvdn-pair-wise-decomposed-value-functions","slug":"pairvdn-pair-wise-decomposed-value-functions","title":"PairVDN - Pair-wise Decomposed Value Functions","date":"2025-03-12","arxiv_id":"2503.09521","repositories_listed":1,"syntology":null},{"url":"/paper/popgym-arcade-parallel-pixelated-pomdps","slug":"popgym-arcade-parallel-pixelated-pomdps","title":"POPGym Arcade: Parallel Pixelated POMDPs","date":"2025-03-03","arxiv_id":"2503.01450","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/popgym-arcade-parallel-pixelated-pomdps#ran","syntology_url":"https://syntology.ai/paper/2503.01450","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.01450"}},"official":{"repos":["bolt-research/popgym_arcade"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/digi-q-learning-q-value-functions-for","slug":"digi-q-learning-q-value-functions-for","title":"Digi-Q: Learning Q-Value Functions for Training Device-Control Agents","date":"2025-02-13","arxiv_id":"2502.15760","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/digi-q-learning-q-value-functions-for#ran","syntology_url":"https://syntology.ai/paper/2502.15760","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.15760"}},"official":{"repos":["digirl-agent/digiq"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/evolution-of-cooperation-in-a-bimodal-mixture","slug":"evolution-of-cooperation-in-a-bimodal-mixture","title":"Evolution of cooperation in a bimodal mixture of conditional cooperators","date":"2025-02-11","arxiv_id":"2502.07537","repositories_listed":1,"syntology":null},{"url":"/paper/conrft-a-reinforced-fine-tuning-method-for","slug":"conrft-a-reinforced-fine-tuning-method-for","title":"ConRFT: A Reinforced Fine-tuning Method for VLA Models via Consistency Policy","date":"2025-02-08","arxiv_id":"2502.05450","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conrft-a-reinforced-fine-tuning-method-for#ran","syntology_url":"https://syntology.ai/paper/2502.05450","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.05450"}},"official":{"repos":["cccedric/conrft"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dual-ensembled-multiagent-q-learning-with","slug":"dual-ensembled-multiagent-q-learning-with","title":"Dual Ensembled Multiagent Q-Learning with Hypernet Regularizer","date":"2025-02-04","arxiv_id":"2502.02018","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-of-deep-reinforcement","slug":"an-empirical-study-of-deep-reinforcement","title":"An Empirical Study of Deep Reinforcement Learning in Continuing Tasks","date":"2025-01-12","arxiv_id":"2501.06937","repositories_listed":1,"syntology":null},{"url":"/paper/decoding-fairness-a-reinforcement-learning","slug":"decoding-fairness-a-reinforcement-learning","title":"Decoding fairness: a reinforcement learning perspective","date":"2024-12-20","arxiv_id":"2412.16249","repositories_listed":1,"syntology":null},{"url":"/paper/maclight-multi-scene-aggregation","slug":"maclight-multi-scene-aggregation","title":"MacLight: Multi-scene Aggregation Convolutional Learning for Traffic Signal Control","date":"2024-12-20","arxiv_id":"2412.15703","repositories_listed":1,"syntology":null},{"url":"/paper/drl4aoi-a-drl-framework-for-semantic-aware","slug":"drl4aoi-a-drl-framework-for-semantic-aware","title":"DRL4AOI: A DRL Framework for Semantic-aware AOI Segmentation in Location-Based Services","date":"2024-12-06","arxiv_id":"2412.05437","repositories_listed":1,"syntology":null},{"url":"/paper/pretrained-llm-adapted-with-lora-as-a","slug":"pretrained-llm-adapted-with-lora-as-a","title":"Pretrained LLM Adapted with LoRA as a Decision Transformer for Offline RL in Quantitative Trading","date":"2024-11-26","arxiv_id":"2411.17900","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-robot-assistive-behaviour-with","slug":"enhancing-robot-assistive-behaviour-with","title":"Enhancing Robot Assistive Behaviour with Reinforcement Learning and Theory of Mind","date":"2024-11-11","arxiv_id":"2411.07003","repositories_listed":1,"syntology":null},{"url":"/paper/think-smart-act-smarl-analyzing-probabilistic","slug":"think-smart-act-smarl-analyzing-probabilistic","title":"Think Smart, Act SMARL! Analyzing Probabilistic Logic Shields for Multi-Agent Reinforcement Learning","date":"2024-11-07","arxiv_id":"2411.04867","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-difference-learning-using","slug":"temporal-difference-learning-using","title":"Temporal-Difference Learning Using Distributed Error Signals","date":"2024-11-06","arxiv_id":"2411.03604","repositories_listed":1,"syntology":{"n":24,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":14,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 14 unverified","sample_list":"/paper/temporal-difference-learning-using#ran","syntology_url":"https://syntology.ai/paper/2411.03604","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.03604"}},"official":{"repos":["social-ai-uoft/ad-paper"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/simulation-of-nanorobots-with-artificial","slug":"simulation-of-nanorobots-with-artificial","title":"Simulation of Nanorobots with Artificial Intelligence and Reinforcement Learning for Advanced Cancer Cell Detection and Tracking","date":"2024-11-04","arxiv_id":"2411.02345","repositories_listed":1,"syntology":null},{"url":"/paper/q-learning-for-quantile-mdps-a-decomposition","slug":"q-learning-for-quantile-mdps-a-decomposition","title":"Q-learning for Quantile MDPs: A Decomposition, Performance, and Convergence Analysis","date":"2024-10-31","arxiv_id":"2410.24128","repositories_listed":1,"syntology":null},{"url":"/paper/zonal-rl-rrt-integrated-rl-rrt-path-planning","slug":"zonal-rl-rrt-integrated-rl-rrt-path-planning","title":"Zonal RL-RRT: Integrated RL-RRT Path Planning with Collision Probability and Zone Connectivity","date":"2024-10-31","arxiv_id":"2410.24205","repositories_listed":1,"syntology":null},{"url":"/paper/q-distribution-guided-q-learning-for-offline","slug":"q-distribution-guided-q-learning-for-offline","title":"Q-Distribution guided Q-learning for offline reinforcement learning: Uncertainty penalized Q-value via consistency model","date":"2024-10-27","arxiv_id":"2410.20312","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/q-distribution-guided-q-learning-for-offline#ran","syntology_url":"https://syntology.ai/paper/2410.20312","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20312"}},"official":{"repos":["evalarzj/qdq"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/streaming-deep-reinforcement-learning-finally","slug":"streaming-deep-reinforcement-learning-finally","title":"Streaming Deep Reinforcement Learning Finally Works","date":"2024-10-18","arxiv_id":"2410.14606","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/streaming-deep-reinforcement-learning-finally#ran","syntology_url":"https://syntology.ai/paper/2410.14606","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14606"}},"official":{"repos":["mohmdelsayed/streaming-drl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-free-world-models-for-online-imitation","slug":"reward-free-world-models-for-online-imitation","title":"Reward-free World Models for Online Imitation Learning","date":"2024-10-17","arxiv_id":"2410.14081","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/reward-free-world-models-for-online-imitation#ran","syntology_url":"https://syntology.ai/paper/2410.14081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14081"}},"official":{"repos":["tobyleelsz/iqmpc"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/a-multi-agent-multi-environment-mixed-q","slug":"a-multi-agent-multi-environment-mixed-q","title":"A Multi-Agent Multi-Environment Mixed Q-Learning for Partially Decentralized Wireless Network Optimization","date":"2024-09-24","arxiv_id":"2409.16450","repositories_listed":1,"syntology":null},{"url":"/paper/audio-driven-reinforcement-learning-for-head","slug":"audio-driven-reinforcement-learning-for-head","title":"Audio-Driven Reinforcement Learning for Head-Orientation in Naturalistic Environments","date":"2024-09-16","arxiv_id":"2409.10048","repositories_listed":1,"syntology":null},{"url":"/paper/offline-reinforcement-learning-for-learning","slug":"offline-reinforcement-learning-for-learning","title":"Offline Reinforcement Learning for Learning to Dispatch for Job Shop Scheduling","date":"2024-09-16","arxiv_id":"2409.10589","repositories_listed":1,"syntology":null},{"url":"/paper/double-successive-over-relaxation-q-learning","slug":"double-successive-over-relaxation-q-learning","title":"Double Successive Over-Relaxation Q-Learning with an Extension to Deep Reinforcement Learning","date":"2024-09-10","arxiv_id":"2409.06356","repositories_listed":1,"syntology":null},{"url":"/paper/robust-q-learning-under-corrupted-rewards","slug":"robust-q-learning-under-corrupted-rewards","title":"Robust Q-Learning under Corrupted Rewards","date":"2024-09-05","arxiv_id":"2409.03237","repositories_listed":1,"syntology":null},{"url":"/paper/crowd-intelligence-for-early-misinformation","slug":"crowd-intelligence-for-early-misinformation","title":"Crowd Intelligence for Early Misinformation Prediction on Social Media","date":"2024-08-08","arxiv_id":"2408.04463","repositories_listed":1,"syntology":null},{"url":"/paper/hard-prompts-made-interpretable-sparse","slug":"hard-prompts-made-interpretable-sparse","title":"Hard Prompts Made Interpretable: Sparse Entropy Regularization for Prompt Tuning with RL","date":"2024-07-20","arxiv_id":"2407.14733","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hard-prompts-made-interpretable-sparse#ran","syntology_url":"https://syntology.ai/paper/2407.14733","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14733"}},"official":{"repos":["youseob/pin"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/2407-21025","slug":"2407-21025","title":"Reinforcement Learning in High-frequency Market Making","date":"2024-07-14","arxiv_id":"2407.21025","repositories_listed":1,"syntology":null},{"url":"/paper/a-two-step-minimax-q-learning-algorithm-for","slug":"a-two-step-minimax-q-learning-algorithm-for","title":"A Multi-Step Minimax Q-learning Algorithm for Two-Player Zero-Sum Markov Games","date":"2024-07-05","arxiv_id":"2407.04240","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-two-step-minimax-q-learning-algorithm-for#ran","syntology_url":"https://syntology.ai/paper/2407.04240","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04240"}},"official":{"repos":["shreyassr123/multi-step-markov-games"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-q-learning-for-finite-ambiguity-sets","slug":"robust-q-learning-for-finite-ambiguity-sets","title":"Robust Q-Learning for finite ambiguity sets","date":"2024-07-05","arxiv_id":"2407.04259","repositories_listed":1,"syntology":null},{"url":"/paper/simplifying-deep-temporal-difference-learning","slug":"simplifying-deep-temporal-difference-learning","title":"Simplifying Deep Temporal Difference Learning","date":"2024-07-05","arxiv_id":"2407.04811","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/simplifying-deep-temporal-difference-learning#ran","syntology_url":"https://syntology.ai/paper/2407.04811","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04811"}},"official":{"repos":["mttga/purejaxql"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/q-adapter-training-your-llm-adapter-as-a","slug":"q-adapter-training-your-llm-adapter-as-a","title":"Q-Adapter: Customizing Pre-trained LLMs to New Preferences with Forgetting Mitigation","date":"2024-07-04","arxiv_id":"2407.03856","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":9,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/q-adapter-training-your-llm-adapter-as-a#ran","syntology_url":"https://syntology.ai/paper/2407.03856","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03856"}},"official":{"repos":["mansicer/Q-Adapter"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/contextualized-hybrid-ensemble-q-learning","slug":"contextualized-hybrid-ensemble-q-learning","title":"Contextualized Hybrid Ensemble Q-learning: Learning Fast with Control Priors","date":"2024-06-28","arxiv_id":"2406.19768","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-soft-q-learning-by-bounding","slug":"boosting-soft-q-learning-by-bounding","title":"Boosting Soft Q-Learning by Bounding","date":"2024-06-26","arxiv_id":"2406.18033","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-routing-for","slug":"reinforcement-learning-based-routing-for","title":"Reinforcement-Learning based routing for packet-optical networks with hybrid telemetry","date":"2024-06-18","arxiv_id":"2406.12602","repositories_listed":1,"syntology":null},{"url":"/paper/probing-implicit-bias-in-semi-gradient-q","slug":"probing-implicit-bias-in-semi-gradient-q","title":"Probing Implicit Bias in Semi-gradient Q-learning: Visualizing the Effective Loss Landscapes via the Fokker--Planck Equation","date":"2024-06-12","arxiv_id":"2406.08148","repositories_listed":1,"syntology":null},{"url":"/paper/plandq-hierarchical-plan-orchestration-via-d","slug":"plandq-hierarchical-plan-orchestration-via-d","title":"PlanDQ: Hierarchical Plan Orchestration via D-Conductor and Q-Performer","date":"2024-06-10","arxiv_id":"2406.06793","repositories_listed":1,"syntology":null},{"url":"/paper/stabilizing-extreme-q-learning-by-maclaurin","slug":"stabilizing-extreme-q-learning-by-maclaurin","title":"Stabilizing Extreme Q-learning by Maclaurin Expansion","date":"2024-06-07","arxiv_id":"2406.04896","repositories_listed":1,"syntology":null},{"url":"/paper/strategically-conservative-q-learning","slug":"strategically-conservative-q-learning","title":"Strategically Conservative Q-Learning","date":"2024-06-06","arxiv_id":"2406.04534","repositories_listed":1,"syntology":null},{"url":"/paper/qroa-a-black-box-query-response-optimization","slug":"qroa-a-black-box-query-response-optimization","title":"Towards Universal and Black-Box Query-Response Only Attack on LLMs with QROA","date":"2024-06-04","arxiv_id":"2406.02044","repositories_listed":1,"syntology":null},{"url":"/paper/target-networks-and-over-parameterization","slug":"target-networks-and-over-parameterization","title":"Target Networks and Over-parameterization Stabilize Off-policy Bootstrapping with Function Approximation","date":"2024-05-31","arxiv_id":"2405.21043","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/target-networks-and-over-parameterization#ran","syntology_url":"https://syntology.ai/paper/2405.21043","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.21043"}},"official":{"repos":["FengdiC/OTTD"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-policies-creating-a-trust-region","slug":"diffusion-policies-creating-a-trust-region","title":"Diffusion Policies creating a Trust Region for Offline Reinforcement Learning","date":"2024-05-30","arxiv_id":"2405.19690","repositories_listed":1,"syntology":{"n":20,"n_ran":17,"n_constructed":0,"n_ran_checked":11,"n_instrument":6,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":20,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/diffusion-policies-creating-a-trust-region#ran","syntology_url":"https://syntology.ai/paper/2405.19690","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19690"}},"official":{"repos":["tianyucodings/diffusion_trusted_q_learning"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/adr-bc-adversarial-density-weighted","slug":"adr-bc-adversarial-density-weighted","title":"Imitating from auxiliary imperfect demonstrations via Adversarial Density Weighted Regression","date":"2024-05-28","arxiv_id":"2405.20351","repositories_listed":1,"syntology":null},{"url":"/paper/aligniql-policy-alignment-in-implicit-q","slug":"aligniql-policy-alignment-in-implicit-q","title":"AlignIQL: Policy Alignment in Implicit Q-Learning through Constrained Optimization","date":"2024-05-28","arxiv_id":"2405.18187","repositories_listed":1,"syntology":null},{"url":"/paper/safe-multi-agent-reinforcement-learning-with","slug":"safe-multi-agent-reinforcement-learning-with","title":"Safe Multi-Agent Reinforcement Learning with Bilevel Optimization in Autonomous Driving","date":"2024-05-28","arxiv_id":"2405.18209","repositories_listed":1,"syntology":null},{"url":"/paper/a-recipe-for-unbounded-data-augmentation-in","slug":"a-recipe-for-unbounded-data-augmentation-in","title":"A Recipe for Unbounded Data Augmentation in Visual Reinforcement Learning","date":"2024-05-27","arxiv_id":"2405.17416","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-recipe-for-unbounded-data-augmentation-in#ran","syntology_url":"https://syntology.ai/paper/2405.17416","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17416"}},"official":{"repos":["aalmuzairee/dmcgb2"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-play-atari-games-using-dueling-q","slug":"learning-to-play-atari-games-using-dueling-q","title":"Learning To Play Atari Games Using Dueling Q-Learning and Hebbian Plasticity","date":"2024-05-22","arxiv_id":"2405.13960","repositories_listed":1,"syntology":null},{"url":"/paper/swiftrl-towards-efficient-reinforcement","slug":"swiftrl-towards-efficient-reinforcement","title":"SwiftRL: Towards Efficient Reinforcement Learning on Real Processing-In-Memory Systems","date":"2024-05-07","arxiv_id":"2405.03967","repositories_listed":1,"syntology":null},{"url":"/paper/regularized-q-learning-through-robust","slug":"regularized-q-learning-through-robust","title":"Regularized Q-learning through Robust Averaging","date":"2024-05-03","arxiv_id":"2405.02201","repositories_listed":1,"syntology":null},{"url":"/paper/afu-actor-free-critic-updates-in-off-policy","slug":"afu-actor-free-critic-updates-in-off-policy","title":"AFU: Actor-Free critic Updates in off-policy RL for continuous control","date":"2024-04-24","arxiv_id":"2404.16159","repositories_listed":1,"syntology":null},{"url":"/paper/research-on-robot-path-planning-based-on","slug":"research-on-robot-path-planning-based-on","title":"Research on Robot Path Planning Based on Reinforcement Learning","date":"2024-04-22","arxiv_id":"2404.14077","repositories_listed":1,"syntology":null},{"url":"/paper/superior-genetic-algorithms-for-the-target","slug":"superior-genetic-algorithms-for-the-target","title":"Superior Genetic Algorithms for the Target Set Selection Problem Based on Power-Law Parameter Choices and Simple Greedy Heuristics","date":"2024-04-05","arxiv_id":"2404.04018","repositories_listed":1,"syntology":null},{"url":"/paper/laser-learning-environment-a-new-environment","slug":"laser-learning-environment-a-new-environment","title":"Laser Learning Environment: A new environment for coordination-critical multi-agent tasks","date":"2024-04-04","arxiv_id":"2404.03596","repositories_listed":1,"syntology":null},{"url":"/paper/from-two-dimensional-to-three-dimensional-1","slug":"from-two-dimensional-to-three-dimensional-1","title":"From Two-Dimensional to Three-Dimensional Environment with Q-Learning: Modeling Autonomous Navigation with Reinforcement Learning and no Libraries","date":"2024-03-27","arxiv_id":"2403.18219","repositories_listed":1,"syntology":null},{"url":"/paper/compressed-federated-reinforcement-learning","slug":"compressed-federated-reinforcement-learning","title":"Compressed Federated Reinforcement Learning with a Generative Model","date":"2024-03-26","arxiv_id":"2404.10635","repositories_listed":1,"syntology":null},{"url":"/paper/a-fairness-oriented-reinforcement-learning","slug":"a-fairness-oriented-reinforcement-learning","title":"A Fairness-Oriented Reinforcement Learning Approach for the Operation and Control of Shared Micromobility Services","date":"2024-03-23","arxiv_id":"2403.15780","repositories_listed":1,"syntology":null},{"url":"/paper/ensembling-prioritized-hybrid-policies-for","slug":"ensembling-prioritized-hybrid-policies-for","title":"Ensembling Prioritized Hybrid Policies for Multi-agent Pathfinding","date":"2024-03-12","arxiv_id":"2403.07559","repositories_listed":1,"syntology":null},{"url":"/paper/scalable-online-exploration-via-coverability","slug":"scalable-online-exploration-via-coverability","title":"Scalable Online Exploration via Coverability","date":"2024-03-11","arxiv_id":"2403.06571","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scalable-online-exploration-via-coverability#ran","syntology_url":"https://syntology.ai/paper/2403.06571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.06571"}},"official":{"repos":["philip-amortila/l1-coverability"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/belief-enriched-pessimistic-q-learning","slug":"belief-enriched-pessimistic-q-learning","title":"Belief-Enriched Pessimistic Q-Learning against Adversarial State Perturbations","date":"2024-03-06","arxiv_id":"2403.04050","repositories_listed":1,"syntology":{"n":24,"n_ran":14,"n_constructed":0,"n_ran_checked":11,"n_instrument":3,"n_unverified":10,"n_honours":0,"n_violates":1,"n_no_contract":10,"n_pointer_only":24,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/belief-enriched-pessimistic-q-learning#ran","syntology_url":"https://syntology.ai/paper/2403.04050","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04050"}},"official":{"repos":["sliencerx/belief-enriched-robust-q-learning"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":10,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-episodic-memory-utilization-of","slug":"efficient-episodic-memory-utilization-of","title":"Efficient Episodic Memory Utilization of Cooperative Multi-Agent Reinforcement Learning","date":"2024-03-02","arxiv_id":"2403.01112","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/efficient-episodic-memory-utilization-of#ran","syntology_url":"https://syntology.ai/paper/2403.01112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01112"}},"official":{"repos":["hyunghona/emu"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/subiq-inverse-soft-q-learning-for-offline","slug":"subiq-inverse-soft-q-learning-for-offline","title":"SPRINQL: Sub-optimal Demonstrations driven Offline Imitation Learning","date":"2024-02-20","arxiv_id":"2402.13147","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/subiq-inverse-soft-q-learning-for-offline#ran","syntology_url":"https://syntology.ai/paper/2402.13147","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13147"}},"official":{"repos":["hmhuy0/SPRINQL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/conservative-and-risk-aware-offline-multi","slug":"conservative-and-risk-aware-offline-multi","title":"Conservative and Risk-Aware Offline Multi-Agent Reinforcement Learning","date":"2024-02-13","arxiv_id":"2402.08421","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/conservative-and-risk-aware-offline-multi#ran","syntology_url":"https://syntology.ai/paper/2402.08421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.08421"}},"official":{"repos":["eslam211/conservative-and-distributional-marl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-digital-cousins-for-ensemble-q","slug":"leveraging-digital-cousins-for-ensemble-q","title":"Leveraging Digital Cousins for Ensemble Q-Learning in Large-Scale Wireless Networks","date":"2024-02-12","arxiv_id":"2402.08022","repositories_listed":1,"syntology":null},{"url":"/paper/solving-deep-reinforcement-learning","slug":"solving-deep-reinforcement-learning","title":"Solving Deep Reinforcement Learning Tasks with Evolution Strategies and Linear Policy Networks","date":"2024-02-10","arxiv_id":"2402.06912","repositories_listed":1,"syntology":null},{"url":"/paper/multi-timescale-ensemble-q-learning-for","slug":"multi-timescale-ensemble-q-learning-for","title":"Multi-Timescale Ensemble Q-learning for Markov Decision Process Policy Optimization","date":"2024-02-08","arxiv_id":"2402.05476","repositories_listed":1,"syntology":null},{"url":"/paper/logical-specifications-guided-dynamic-task","slug":"logical-specifications-guided-dynamic-task","title":"Logical Specifications-guided Dynamic Task Sampling for Reinforcement Learning Agents","date":"2024-02-06","arxiv_id":"2402.03678","repositories_listed":1,"syntology":null},{"url":"/paper/raddqn-a-deep-q-learning-based-architecture","slug":"raddqn-a-deep-q-learning-based-architecture","title":"RadDQN: a Deep Q Learning-based Architecture for Finding Time-efficient Minimum Radiation Exposure Pathway","date":"2024-02-01","arxiv_id":"2402.00468","repositories_listed":1,"syntology":null},{"url":"/paper/information-theoretic-state-variable","slug":"information-theoretic-state-variable","title":"Information-Theoretic State Variable Selection for Reinforcement Learning","date":"2024-01-21","arxiv_id":"2401.11512","repositories_listed":1,"syntology":null},{"url":"/paper/vqc-based-reinforcement-learning-with-data-re","slug":"vqc-based-reinforcement-learning-with-data-re","title":"VQC-Based Reinforcement Learning with Data Re-uploading: Performance and Trainability","date":"2024-01-21","arxiv_id":"2401.11555","repositories_listed":1,"syntology":null},{"url":"/paper/a-semantic-aware-multiple-access-scheme-for","slug":"a-semantic-aware-multiple-access-scheme-for","title":"A Semantic-Aware Multiple Access Scheme for Distributed, Dynamic 6G-Based Applications","date":"2024-01-12","arxiv_id":"2401.06308","repositories_listed":1,"syntology":null},{"url":"/paper/decision-making-in-non-stationary-1","slug":"decision-making-in-non-stationary-1","title":"Decision Making in Non-Stationary Environments with Policy-Augmented Search","date":"2024-01-06","arxiv_id":"2401.03197","repositories_listed":1,"syntology":null},{"url":"/paper/spqr-controlling-q-ensemble-independence-with-1","slug":"spqr-controlling-q-ensemble-independence-with-1","title":"SPQR: Controlling Q-ensemble Independence with Spiked Random Model for Reinforcement Learning","date":"2024-01-06","arxiv_id":"2401.03137","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/spqr-controlling-q-ensemble-independence-with-1#ran","syntology_url":"https://syntology.ai/paper/2401.03137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.03137"}},"official":{"repos":["dohyeoklee/SPQR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/investigating-the-performance-and-reliability","slug":"investigating-the-performance-and-reliability","title":"Investigating the Performance and Reliability, of the Q-Learning Algorithm in Various Unknown Environments","date":"2023-12-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-reinforcement-learning-with-4","slug":"sample-efficient-reinforcement-learning-with-4","title":"Sample Efficient Reinforcement Learning with Partial Dynamics Knowledge","date":"2023-12-19","arxiv_id":"2312.12558","repositories_listed":1,"syntology":null},{"url":"/paper/i-open-at-the-close-a-deep-reinforcement","slug":"i-open-at-the-close-a-deep-reinforcement","title":"I Open at the Close: A Deep Reinforcement Learning Evaluation of Open Streets Initiatives","date":"2023-12-12","arxiv_id":"2312.07680","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-sparse-reward-goal-conditioned","slug":"efficient-sparse-reward-goal-conditioned","title":"Efficient Sparse-Reward Goal-Conditioned Reinforcement Learning with a High Replay Ratio and Regularization","date":"2023-12-10","arxiv_id":"2312.05787","repositories_listed":1,"syntology":null},{"url":"/paper/synthesis-of-temporally-robust-policies-for","slug":"synthesis-of-temporally-robust-policies-for","title":"Synthesis of Temporally-Robust Policies for Signal Temporal Logic Tasks using Reinforcement Learning","date":"2023-12-10","arxiv_id":"2312.05764","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-via-1","slug":"multi-agent-reinforcement-learning-via-1","title":"Multi-Agent Reinforcement Learning via Distributed MPC as a Function Approximator","date":"2023-12-08","arxiv_id":"2312.05166","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-distributed-reinforcement-learning","slug":"optimizing-distributed-reinforcement-learning","title":"Efficient Parallel Reinforcement Learning Framework using the Reactor Model","date":"2023-12-07","arxiv_id":"2312.04704","repositories_listed":1,"syntology":null},{"url":"/paper/l-m-v-iql-multiple-intention-inverse","slug":"l-m-v-iql-multiple-intention-inverse","title":"Multi-intention Inverse Q-learning for Interpretable Behavior Representation","date":"2023-11-23","arxiv_id":"2311.13870","repositories_listed":1,"syntology":null},{"url":"/paper/optimistic-multi-agent-policy-gradient-for","slug":"optimistic-multi-agent-policy-gradient-for","title":"Optimistic Multi-Agent Policy Gradient","date":"2023-11-03","arxiv_id":"2311.01953","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimistic-multi-agent-policy-gradient-for#ran","syntology_url":"https://syntology.ai/paper/2311.01953","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.01953"}},"official":{"repos":["wenshuaizhao/optimappo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/free-from-bellman-completeness-trajectory","slug":"free-from-bellman-completeness-trajectory","title":"Free from Bellman Completeness: Trajectory Stitching via Model-based Return-conditioned Supervised Learning","date":"2023-10-30","arxiv_id":"2310.19308","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/free-from-bellman-completeness-trajectory#ran","syntology_url":"https://syntology.ai/paper/2310.19308","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19308"}},"official":{"repos":["zhaoyizhou1123/mbrcsl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/deep-reinforcement-learning-based-intelligent-2","slug":"deep-reinforcement-learning-based-intelligent-2","title":"Deep Reinforcement Learning-based Intelligent Traffic Signal Controls with Optimized CO2 emissions","date":"2023-10-19","arxiv_id":"2310.13129","repositories_listed":1,"syntology":null},{"url":"/paper/learning-rl-policies-for-joint-beamforming","slug":"learning-rl-policies-for-joint-beamforming","title":"Learning RL-Policies for Joint Beamforming Without Exploration: A Batch Constrained Off-Policy Approach","date":"2023-10-12","arxiv_id":"2310.08660","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-continuous-control-with-consistency","slug":"boosting-continuous-control-with-consistency","title":"Boosting Continuous Control with Consistency Policy","date":"2023-10-10","arxiv_id":"2310.06343","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/boosting-continuous-control-with-consistency#ran","syntology_url":"https://syntology.ai/paper/2310.06343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.06343"}},"official":{"repos":["cccedric/cpql"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deepqtest-testing-autonomous-driving-systems","slug":"deepqtest-testing-autonomous-driving-systems","title":"DeepQTest: Testing Autonomous Driving Systems with Reinforcement Learning and Real-world Weather Data","date":"2023-10-08","arxiv_id":"2310.05170","repositories_listed":1,"syntology":null}],"record_sha256":"6ff55afd208cc17f7a88b496f00f9ad8baf98f4a48564e754f3786fb3a7c5b86","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}