{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/reinforcement-learning-for-optimization-of","title":"Reinforcement Learning for Optimization of COVID-19 Mitigation policies","arxiv_id":"2010.10560","date":"2020-10-20","proceeding":null,"authors":["Varun Kompella","Roberto Capobianco","Stacy Jong","Jonathan Browne","Spencer Fox","Lauren Meyers","Peter Wurman","Peter Stone"],"abstract":"The year 2020 has seen the COVID-19 virus lead to one of the worst global pandemics in history. As a result, governments around the world are faced with the challenge of protecting public health, while keeping the economy running to the greatest extent possible. Epidemiological models provide insight into the spread of these types of diseases and predict the effects of possible intervention policies. However, to date,the even the most data-driven intervention policies rely on heuristics. In this paper, we study how reinforcement learning (RL) can be used to optimize mitigation policies that minimize the economic impact without overwhelming the hospital capacity. Our main contributions are (1) a novel agent-based pandemic simulator which, unlike traditional models, is able to model fine-grained interactions among people at specific locations in a community; and (2) an RL-based methodology for optimizing fine-grained mitigation policies within this simulator. Our results validate both the overall simulator behavior and the learned policies under realistic conditions.","url_abs":"https://arxiv.org/abs/2010.10560v1","url_pdf":"https://arxiv.org/pdf/2010.10560v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"reinforcement-learning-for-optimization-of","repo_url":"https://github.com/SonyAI/PandemicSimulator","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"none","reach":{"status":"ok","spdx":"Apache-2.0"}}],"tasks":[{"task_slug":"reinforcement-learning","task_name":"Reinforcement Learning"},{"task_slug":"reinforcement-learning-1","task_name":"Reinforcement Learning (RL)"},{"task_slug":"reinforcement-learning-2","task_name":"reinforcement-learning"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2010.10560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.10560"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/SonyAI/PandemicSimulator","reach":{"status":"ok","spdx":"Apache-2.0"}}],"summary":{"unverified":3},"by_repo_kind":{"official":{"samples":3,"ran":0,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"534fae7058b4859b","entry":"cluster_into_random_sized_groups","repo":"SonyAI/PandemicSimulator","repo_kind":"official","path":"python/pandemic_simulator/utils.py","file_url":"https://github.com/SonyAI/PandemicSimulator/blob/HEAD/python/pandemic_simulator/utils.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"534fae7058b4859b"}},{"code_sha256_prefix":"bcad497509ee1d53","entry":"get_us_age_distribution","repo":"SonyAI/PandemicSimulator","repo_kind":"official","path":"python/pandemic_simulator/environment/make_population.py","file_url":"https://github.com/SonyAI/PandemicSimulator/blob/HEAD/python/pandemic_simulator/environment/make_population.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"bcad497509ee1d53"}},{"code_sha256_prefix":"9936f62058d64b50","entry":"shallow_asdict","repo":"SonyAI/PandemicSimulator","repo_kind":"official","path":"python/pandemic_simulator/utils.py","file_url":"https://github.com/SonyAI/PandemicSimulator/blob/HEAD/python/pandemic_simulator/utils.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"9936f62058d64b50"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}