{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/q-learning/papers/20","list_of":"/task/q-learning","task":"Q-Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":20,"pages_in_order":20,"rows_per_page":100,"rows":[1901,1918],"of":1918,"counts":{"archive_papers_tagged":1918,"with_a_code_link":463,"where_syntology_ran_a_sample":119,"not_listed_spam_title":0,"listed":1918,"listed_where_code_ran":119,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":102,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":102,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/q-learning","prev":"/task/q-learning/papers/19","next":null,"papers":[{"url":null,"slug":"q-learning-for-optimal-control-of-continuous","title":"Q-learning for Optimal Control of Continuous-time Systems","date":"2014-10-11","arxiv_id":"1410.2954","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-cooperate-via-policy-search","title":"Learning to Cooperate via Policy Search","date":"2014-08-07","arxiv_id":"1408.1484","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-algorithm-for","title":"Reinforcement Learning Based Algorithm for the Maximization of EV Charging Station Revenue","date":"2014-07-04","arxiv_id":"1407.1291","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-medical-treatments-using-novel","title":"Personalized Medical Treatments Using Novel Reinforcement Learning Algorithms","date":"2014-06-16","arxiv_id":"1406.3922","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-agent-vs-multi-agent-techniques-for","title":"Single-Agent vs. Multi-Agent Techniques for Concurrent Reinforcement Learning of Negotiation Dialogue Policies","date":"2014-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"empirically-evaluating-multiagent-learning","title":"Empirically Evaluating Multiagent Learning Algorithms","date":"2014-01-31","arxiv_id":"1401.8074","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-stochastic-resource-control-a","title":"Adaptive Stochastic Resource Control: A Machine Learning Approach","date":"2014-01-15","arxiv_id":"1401.3434","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-demand-response-using-device-based","title":"Optimal Demand Response Using Device Based Reinforcement Learning","date":"2014-01-08","arxiv_id":"1401.1549","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-timescale-convergent-q-learning-for-sleep","title":"Two Timescale Convergent Q-learning for Sleep--Scheduling in Wireless Sensor Networks","date":"2013-12-27","arxiv_id":"1312.7292","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-optimization-in-a-multi-agents","title":"Q-learning optimization in a multi-agents system for image segmentation","date":"2013-11-23","arxiv_id":"1311.6054","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-sensitive-reinforcement-learning","title":"Risk-sensitive Reinforcement Learning","date":"2013-11-08","arxiv_id":"1311.2097","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximate-kalman-filter-q-learning-for","title":"Approximate Kalman Filter Q-Learning for Continuous State-Space MDPs","date":"2013-09-26","arxiv_id":"1309.6868","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-association-problem-in-wireless-networks","title":"The association problem in wireless networks: a Policy Gradient Reinforcement Learning approach","date":"2013-06-11","arxiv_id":"1306.2554","repositories_listed":0,"syntology":null},{"url":null,"slug":"projective-simulation-for-classical-learning","title":"Projective simulation for classical learning agents: a comprehensive investigation","date":"2013-05-07","arxiv_id":"1305.1578","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-q-learning-applied-to-ubiquitous","title":"Hybrid Q-Learning Applied to Ubiquitous recommender system","date":"2013-03-10","arxiv_id":"1303.2651","repositories_listed":0,"syntology":null},{"url":null,"slug":"speedy-q-learning","title":"Speedy Q-Learning","date":"2011-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/double-q-learning","slug":"double-q-learning","title":"Double Q-learning","date":"2010-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"convergent-temporal-difference-learning-with","title":"Convergent Temporal-Difference Learning with Arbitrary Smooth Function Approximation","date":"2009-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"1b8dd72d75c1b7ea6cdf2809d083135beae640b9b4a28a65dfde0d4a86a82a80","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}