{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/end-to-end-model-free-reinforcement-learning","title":"End-to-End Model-Free Reinforcement Learning for Urban Driving using Implicit Affordances","arxiv_id":"1911.10868","date":"2019-11-25","proceeding":"CVPR 2020 6","authors":["Marin Toromanoff","Emilie Wirbel","Fabien Moutarde"],"abstract":"Reinforcement Learning (RL) aims at learning an optimal behavior policy from its own experiments and not rule-based control methods. However, there is no RL algorithm yet capable of handling a task as difficult as urban driving. We present a novel technique, coined implicit affordances, to effectively leverage RL for urban driving thus including lane keeping, pedestrians and vehicles avoidance, and traffic light detection. To our knowledge we are the first to present a successful RL agent handling such a complex task especially regarding the traffic light detection. Furthermore, we have demonstrated the effectiveness of our method by winning the Camera Only track of the CARLA challenge.","url_abs":"https://arxiv.org/abs/1911.10868v2","url_pdf":"https://arxiv.org/pdf/1911.10868v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"end-to-end-model-free-reinforcement-learning","repo_url":"https://github.com/valeoai/LearningByCheating","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}}],"tasks":[{"task_slug":"autonomous-driving","task_name":"Autonomous Driving"},{"task_slug":"reinforcement-learning","task_name":"Reinforcement Learning"},{"task_slug":"reinforcement-learning-1","task_name":"Reinforcement Learning (RL)"},{"task_slug":"reinforcement-learning-2","task_name":"reinforcement-learning"}],"methods":[{"method_slug":"carla","method_name":"CARLA"},{"method_slug":"entropy-regularization","method_name":"Entropy Regularization"},{"method_slug":"ppo","method_name":"PPO"}],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/autonomous-driving-on-carla-leaderboard","task":"Autonomous Driving","dataset":"CARLA Leaderboard","model":"MaRLn","rank_in_archive_order":14,"of":18,"metrics":{"Driving Score":"24.98","Infraction penalty":"0.52","Route Completion":"46.97"},"uses_additional_data":false}],"syntology":{"syntology_url":"https://syntology.ai/paper/1911.10868","atlas_url":"https://app.syntology.ai/?focus=1911.10868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.10868"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/valeoai/LearningByCheating","reach":{"status":"ok","spdx":"MIT"}}],"summary":{"ran":1,"ran_draft_wrong":1,"unverified":5},"by_repo_kind":{"official":{"samples":7,"ran":2,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"ccc25780948874cb","entry":"clamp","repo":"valeoai/LearningByCheating","repo_kind":"official","path":"misc/dynamic_weather.py","file_url":"https://github.com/valeoai/LearningByCheating/blob/HEAD/misc/dynamic_weather.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"ccc25780948874cb"}},{"code_sha256_prefix":"c882c504d11f2b36","entry":"get_actor_display_name","repo":"valeoai/LearningByCheating","repo_kind":"official","path":"misc/automatic_control.py","file_url":"https://github.com/valeoai/LearningByCheating/blob/HEAD/misc/automatic_control.py","link_basis":"plan_row","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"c882c504d11f2b36"}},{"code_sha256_prefix":"ad2b8a18da38cd81","entry":"create_resnet_basic_block","repo":"valeoai/LearningByCheating","repo_kind":"official","path":"bird_view/models/model_supervised.py","file_url":"https://github.com/valeoai/LearningByCheating/blob/HEAD/bird_view/models/model_supervised.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"ad2b8a18da38cd81"}},{"code_sha256_prefix":"663c3ff922453f02","entry":"from_file","repo":"valeoai/LearningByCheating","repo_kind":"official","path":"benchmark/goal_suite.py","file_url":"https://github.com/valeoai/LearningByCheating/blob/HEAD/benchmark/goal_suite.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"663c3ff922453f02"}},{"code_sha256_prefix":"b72f5568cea89800","entry":"get_collision","repo":"valeoai/LearningByCheating","repo_kind":"official","path":"misc/find_traffic_violations.py","file_url":"https://github.com/valeoai/LearningByCheating/blob/HEAD/misc/find_traffic_violations.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"b72f5568cea89800"}},{"code_sha256_prefix":"2b60432345a2845a","entry":"get_town","repo":"valeoai/LearningByCheating","repo_kind":"official","path":"misc/find_traffic_violations.py","file_url":"https://github.com/valeoai/LearningByCheating/blob/HEAD/misc/find_traffic_violations.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"2b60432345a2845a"}},{"code_sha256_prefix":"c0f945e63447e7a1","entry":"parse","repo":"valeoai/LearningByCheating","repo_kind":"official","path":"misc/find_traffic_violations.py","file_url":"https://github.com/valeoai/LearningByCheating/blob/HEAD/misc/find_traffic_violations.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"c0f945e63447e7a1"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}