{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/solving-the-rubiks-cube-without-human","title":"Solving the Rubik's Cube Without Human Knowledge","arxiv_id":"1805.07470","date":"2018-05-18","proceeding":null,"authors":["Stephen McAleer","Forest Agostinelli","Alexander Shmakov","Pierre Baldi"],"abstract":"A generally intelligent agent must be able to teach itself how to solve\nproblems in complex domains with minimal human supervision. Recently, deep\nreinforcement learning algorithms combined with self-play have achieved\nsuperhuman proficiency in Go, Chess, and Shogi without human data or domain\nknowledge. In these environments, a reward is always received at the end of the\ngame, however, for many combinatorial optimization environments, rewards are\nsparse and episodes are not guaranteed to terminate. We introduce Autodidactic\nIteration: a novel reinforcement learning algorithm that is able to teach\nitself how to solve the Rubik's Cube with no human assistance. Our algorithm is\nable to solve 100% of randomly scrambled cubes while achieving a median solve\nlength of 30 moves -- less than or equal to solvers that employ human domain\nknowledge.","url_abs":"http://arxiv.org/abs/1805.07470v1","url_pdf":"http://arxiv.org/pdf/1805.07470v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"solving-the-rubiks-cube-without-human","repo_url":"https://github.com/AashrayAnand/rubiks-cube-reinforcement-learning","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":null},{"paper_slug":"solving-the-rubiks-cube-without-human","repo_url":"https://github.com/Dalkio/RL_rubiks","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok"}},{"paper_slug":"solving-the-rubiks-cube-without-human","repo_url":"https://github.com/JasperBusschers/multi-objective-Rubik-s-cube","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"solving-the-rubiks-cube-without-human","repo_url":"https://github.com/KeatonMueller/rl-cube","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"solving-the-rubiks-cube-without-human","repo_url":"https://github.com/kaletap/deep-cube-rl","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"solving-the-rubiks-cube-without-human","repo_url":"https://github.com/mkovalski/cs4995_cube","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null},{"paper_slug":"solving-the-rubiks-cube-without-human","repo_url":"https://github.com/nascarsayan/dl-rubiks-autodidactic-solver","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok"}},{"paper_slug":"solving-the-rubiks-cube-without-human","repo_url":"https://github.com/nathangrinsztajn/rubiks_cube","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok"}},{"paper_slug":"solving-the-rubiks-cube-without-human","repo_url":"https://github.com/robbiejones96/RubiksSolver","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok"}}],"tasks":[{"task_slug":"combinatorial-optimization","task_name":"Combinatorial Optimization"},{"task_slug":"deep-reinforcement-learning","task_name":"Deep Reinforcement Learning"},{"task_slug":"reinforcement-learning","task_name":"Reinforcement Learning"},{"task_slug":"reinforcement-learning-1","task_name":"Reinforcement Learning (RL)"},{"task_slug":"rubik-s-cube","task_name":"Rubik's Cube"},{"task_slug":"reinforcement-learning-2","task_name":"reinforcement-learning"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1805.07470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.07470"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Dalkio/RL_rubiks","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/JasperBusschers/multi-objective-Rubik-s-cube","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/nathangrinsztajn/rubiks_cube","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/mkovalski/cs4995_cube","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/nascarsayan/dl-rubiks-autodidactic-solver","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/KeatonMueller/rl-cube","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/robbiejones96/RubiksSolver","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/AashrayAnand/rubiks-cube-reinforcement-learning","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/kaletap/deep-cube-rl","reach":{"status":"ok"}}],"summary":{"ran_violates":1,"ran_honours":3},"by_repo_kind":{"listed":{"samples":4,"ran":4,"repositories":2}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":4,"samples":[{"code_sha256_prefix":"dd10cc6cef9c45ec","entry":"move","repo":"mkovalski/cs4995_cube","repo_kind":"listed","path":"adi.py","file_url":"https://github.com/mkovalski/cs4995_cube/blob/HEAD/adi.py","link_basis":"first_harvest_node","language":"python","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"dd10cc6cef9c45ec"}},{"code_sha256_prefix":"01f83b840ef3a509","entry":"num_crosses","repo":"AashrayAnand/rubiks-cube-reinforcement-learning","repo_kind":"listed","path":"puzzle.py","file_url":"https://github.com/AashrayAnand/rubiks-cube-reinforcement-learning/blob/HEAD/puzzle.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"01f83b840ef3a509"}},{"code_sha256_prefix":"f56128e0e5a7b5df","entry":"num_pieces_correct_side","repo":"AashrayAnand/rubiks-cube-reinforcement-learning","repo_kind":"listed","path":"puzzle.py","file_url":"https://github.com/AashrayAnand/rubiks-cube-reinforcement-learning/blob/HEAD/puzzle.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"f56128e0e5a7b5df"}},{"code_sha256_prefix":"7d39dad46d604bd1","entry":"num_solved_sides","repo":"AashrayAnand/rubiks-cube-reinforcement-learning","repo_kind":"listed","path":"puzzle.py","file_url":"https://github.com/AashrayAnand/rubiks-cube-reinforcement-learning/blob/HEAD/puzzle.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"7d39dad46d604bd1"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}