{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/constrained-episodic-reinforcement-learning","title":"Constrained episodic reinforcement learning in concave-convex and knapsack settings","arxiv_id":"2006.05051","date":"2020-06-09","proceeding":"NeurIPS 2020 12","authors":["Kianté Brantley","Miroslav Dudik","Thodoris Lykouris","Sobhan Miryoosefi","Max Simchowitz","Aleksandrs Slivkins","Wen Sun"],"abstract":"We propose an algorithm for tabular episodic reinforcement learning with constraints. We provide a modular analysis with strong theoretical guarantees for settings with concave rewards and convex constraints, and for settings with hard constraints (knapsacks). Most of the previous work in constrained reinforcement learning is limited to linear constraints, and the remaining work focuses on either the feasibility question or settings with a single episode. Our experiments demonstrate that the proposed algorithm significantly outperforms these approaches in existing constrained episodic environments.","url_abs":"https://arxiv.org/abs/2006.05051v2","url_pdf":"https://arxiv.org/pdf/2006.05051v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"constrained-episodic-reinforcement-learning","repo_url":"https://github.com/miryoosefi/ConRL","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"none","reach":null}],"tasks":[{"task_slug":"reinforcement-learning","task_name":"Reinforcement Learning"},{"task_slug":"reinforcement-learning-1","task_name":"Reinforcement Learning (RL)"},{"task_slug":"reinforcement-learning-2","task_name":"reinforcement-learning"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2006.05051","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.05051"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/miryoosefi/ConRL","reach":null}],"summary":{"ran":2},"by_repo_kind":{"official":{"samples":2,"ran":2,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":2,"samples":[{"code_sha256_prefix":"96359981f715b7d1","entry":"Appropo","repo":"miryoosefi/ConRL","repo_kind":"official","path":"rlwithknapsacks/src/alg/appropo_og.py","file_url":"https://github.com/miryoosefi/ConRL/blob/HEAD/rlwithknapsacks/src/alg/appropo_og.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"96359981f715b7d1"}},{"code_sha256_prefix":"b17b57e041138529","entry":"MixturePolicy","repo":"miryoosefi/ConRL","repo_kind":"official","path":"rlwithknapsacks/src/alg/appropo_og.py","file_url":"https://github.com/miryoosefi/ConRL/blob/HEAD/rlwithknapsacks/src/alg/appropo_og.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"b17b57e041138529"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}