{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,6]],"date-time":"2026-08-06T19:01:58Z","timestamp":1786042918270,"version":"3.56.0"},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"8","license":[{"start":{"date-parts":[[2019,7,15]],"date-time":"2019-07-15T00:00:00Z","timestamp":1563148800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2019,7,15]],"date-time":"2019-07-15T00:00:00Z","timestamp":1563148800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Nat Mach Intell"],"DOI":"10.1038\/s42256-019-0070-z","type":"journal-article","created":{"date-parts":[[2019,7,15]],"date-time":"2019-07-15T16:03:18Z","timestamp":1563206598000},"page":"356-363","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":93,"title":["Solving the Rubik\u2019s cube with deep reinforcement learning and search"],"prefix":"10.1038","volume":"1","author":[{"given":"Forest","family":"Agostinelli","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Stephen","family":"McAleer","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Alexander","family":"Shmakov","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8752-4664","authenticated-orcid":false,"given":"Pierre","family":"Baldi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2019,7,15]]},"reference":[{"key":"70_CR1","doi-asserted-by":"crossref","unstructured":"Lichodzijewski, P. & Heywood, M. in Genetic Programming Theory and Practice VIII (eds Riolo, R., McConaghy, T. & Vladislavleva, E.) 35\u201354 (Springer, 2011).","DOI":"10.1007\/978-1-4419-7747-2_3"},{"key":"70_CR2","doi-asserted-by":"crossref","unstructured":"Smith, R. J., Kelly, S. & Heywood, M. I. Discovering Rubik\u2019s cube subgroups using coevolutionary GP: a five twist experiment. In Proceedings of the Genetic and Evolutionary Computation Conference 2016 789\u2013796 (ACM, 2016).","DOI":"10.1145\/2908812.2908887"},{"key":"70_CR3","first-page":"57","volume":"1885","author":"R Brunetto","year":"2017","unstructured":"Brunetto, R. & Trunda, O. Deep heuristic-learning in the Rubik\u2019s cube domain: an experimental evaluation. Proc. ITAT 1885, 57\u201364 (2017).","journal-title":"Proc. ITAT"},{"key":"70_CR4","doi-asserted-by":"crossref","unstructured":"Johnson, C. G. Solving the Rubik\u2019s cube with learned guidance functions. In Proceedings of 2018 IEEE Symposium Series on Computational Intelligence (SSCI) 2082\u20132089 (IEEE, 2018).","DOI":"10.1109\/SSCI.2018.8628626"},{"key":"70_CR5","doi-asserted-by":"publisher","first-page":"35","DOI":"10.1016\/0004-3702(85)90012-8","volume":"26","author":"RE Korf","year":"1985","unstructured":"Korf, R. E. Macro-operators: a weak method for learning. Artif. Intell. 26, 35\u201377 (1985).","journal-title":"Artif. Intell."},{"key":"70_CR6","doi-asserted-by":"publisher","first-page":"2075","DOI":"10.1016\/j.artint.2011.08.001","volume":"175","author":"SJ Arfaee","year":"2011","unstructured":"Arfaee, S. J., Zilles, S. & Holte, R. C. Learning heuristic functions for large state spaces. Artif. Intell. 175, 2075\u20132098 (2011).","journal-title":"Artif. Intell."},{"key":"70_CR7","unstructured":"Korf, R. E. Finding optimal solutions to Rubik\u2019s cube using pattern databases. In Proceedings of the Fourteenth National Conference on Artificial Intelligence and Ninth Conference on Innovative Applications of Artificial Intelligence 700\u2013705 (AAAI Press, 1997); http:\/\/dl.acm.org\/citation.cfm?id=1867406.1867515"},{"key":"70_CR8","doi-asserted-by":"publisher","first-page":"9","DOI":"10.1016\/S0004-3702(01)00092-3","volume":"134","author":"RE Korf","year":"2002","unstructured":"Korf, R. E. & Felner, A. Disjoint pattern database heuristics. Artif. Intell. 134, 9\u201322 (2002).","journal-title":"Artif. Intell."},{"key":"70_CR9","doi-asserted-by":"publisher","first-page":"279","DOI":"10.1613\/jair.1480","volume":"22","author":"A Felner","year":"2004","unstructured":"Felner, A., Korf, R. E. & Hanan, S. Additive pattern database heuristics. J. Artif. Intell. Res. 22, 279\u2013318 (2004).","journal-title":"J. Artif. Intell. Res."},{"key":"70_CR10","doi-asserted-by":"publisher","first-page":"5","DOI":"10.1016\/S0004-3702(01)00108-4","volume":"129","author":"B Bonet","year":"2001","unstructured":"Bonet, B. & Geffner, H. Planning as heuristic search. Artif. Intell. 129, 5\u201333 (2001).","journal-title":"Artif. Intell."},{"key":"70_CR11","doi-asserted-by":"publisher","first-page":"85","DOI":"10.1016\/j.neunet.2014.09.003","volume":"61","author":"J Schmidhuber","year":"2015","unstructured":"Schmidhuber, J. Deep learning in neural networks: an overview. Neural Netw. 61, 85\u2013117 (2015).","journal-title":"Neural Netw."},{"key":"70_CR12","unstructured":"Goodfellow, I., Bengio, Y., Courville, A. & Bengio, Y. Deep Learning Vol. 1 (MIT Press, 2016)."},{"key":"70_CR13","unstructured":"Sutton, R. S. & Barto, A. G. Reinforcement Learning: An Introduction Vol. 1 (MIT Press, 1998)."},{"key":"70_CR14","unstructured":"Bellman, R. Dynamic Programming (Princeton Univ. Press, 1957)."},{"key":"70_CR15","doi-asserted-by":"publisher","first-page":"1127","DOI":"10.1287\/mnsc.24.11.1127","volume":"24","author":"ML Puterman","year":"1978","unstructured":"Puterman, M. L. & Shin, M. C. Modified policy iteration algorithms for discounted Markov decision problems. Manage. Sci. 24, 1127\u20131137 (1978).","journal-title":"Manage. Sci."},{"key":"70_CR16","unstructured":"Bertsekas, D. P. & Tsitsiklis, J. N. Neuro-dynamic Programming (Athena Scientific, 1996)."},{"key":"70_CR17","doi-asserted-by":"publisher","first-page":"100","DOI":"10.1109\/TSSC.1968.300136","volume":"4","author":"PE Hart","year":"1968","unstructured":"Hart, P. E., Nilsson, N. J. & Raphael, B. A formal basis for the heuristic determination of minimum cost paths. IEEE Trans. Syst. Sci. Cybern. 4, 100\u2013107 (1968).","journal-title":"IEEE Trans. Syst. Sci. Cybern."},{"key":"70_CR18","doi-asserted-by":"publisher","first-page":"193","DOI":"10.1016\/0004-3702(70)90007-X","volume":"1","author":"I Pohl","year":"1970","unstructured":"Pohl, I. Heuristic search viewed as path finding in a graph. Artif. Intell. 1, 193\u2013204 (1970).","journal-title":"Artif. Intell."},{"key":"70_CR19","doi-asserted-by":"publisher","first-page":"1310","DOI":"10.1016\/j.artint.2009.06.004","volume":"173","author":"R Ebendt","year":"2009","unstructured":"Ebendt, R. & Drechsler, R. Weighted A* search\u2014unifying view and application. Artif. Intell. 173, 1310\u20131342 (2009).","journal-title":"Artif. Intell."},{"key":"70_CR20","unstructured":"McAleer, S., Agostinelli, F., Shmakov, A. & Baldi, P. Solving the Rubik\u2019s cube with approximate policy iteration. Proceedings of International Conference on Learning Representations (ICLR) (PMLR, 2019)."},{"key":"70_CR21","doi-asserted-by":"publisher","first-page":"1140","DOI":"10.1126\/science.aar6404","volume":"362","author":"D Silver","year":"2018","unstructured":"Silver, D. et al. A general reinforcement learning algorithm that masters chess, shogi and Go through self-play. Science 362, 1140\u20131144 (2018).","journal-title":"Science"},{"key":"70_CR22","unstructured":"Rokicki, T. God\u2019s Number is 26 in the Quarter-turn Metric http:\/\/www.cube20.org\/qtm\/ (2014)."},{"key":"70_CR23","doi-asserted-by":"publisher","first-page":"97","DOI":"10.1016\/0004-3702(85)90084-0","volume":"27","author":"RE Korf","year":"1985","unstructured":"Korf, R. E. Depth-first iterative-deepening: an optimal admissible tree search. Artif. Intell. 27, 97\u2013109 (1985).","journal-title":"Artif. Intell."},{"key":"70_CR24","unstructured":"Rokicki, T. cube20 https:\/\/github.com\/rokicki\/cube20src (2016)."},{"key":"70_CR25","doi-asserted-by":"publisher","first-page":"645","DOI":"10.1137\/140973499","volume":"56","author":"T Rokicki","year":"2014","unstructured":"Rokicki, T., Kociemba, H., Davidson, M. & Dethridge, J. The diameter of the Rubik\u2019s cube group is twenty. SIAM Rev. 56, 645\u2013670 (2014).","journal-title":"SIAM Rev."},{"key":"70_CR26","doi-asserted-by":"publisher","first-page":"318","DOI":"10.1111\/0824-7935.00065","volume":"14","author":"JC Culberson","year":"1998","unstructured":"Culberson, J. C. & Schaeffer, J. Pattern databases. Comput. Intell. 14, 318\u2013334 (1998).","journal-title":"Comput. Intell."},{"key":"70_CR27","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S. & Sun, J. Deep residual learning for image recognition. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition 770\u2013778 (IEEE, 2016).","DOI":"10.1109\/CVPR.2016.90"},{"key":"70_CR28","unstructured":"Kociemba, H. 15-Puzzle Optimal Solver http:\/\/kociemba.org\/themen\/fifteen\/fifteensolver.html (2018)."},{"key":"70_CR29","unstructured":"Scherphuis, J. The Mathematics of Lights Out https:\/\/www.jaapsch.net\/puzzles\/lomath.htm (2015)."},{"key":"70_CR30","doi-asserted-by":"publisher","first-page":"215","DOI":"10.1016\/S0925-7721(99)00017-6","volume":"13","author":"D Dor","year":"1999","unstructured":"Dor, D. & Zwick, U. Sokoban and other motion planning problems. Comput. Geom. 13, 215\u2013228 (1999).","journal-title":"Comput. Geom."},{"key":"70_CR31","unstructured":"Guez, A. et al. An Investigation of Model-free Planning: Boxoban Levels https:\/\/github.com\/deepmind\/boxoban-levels\/ (2018)."},{"key":"70_CR32","unstructured":"Orseau, L., Lelis, L., Lattimore, T. & Weber, T. Single-agent policy tree search with guarantees. In Advances in Neural Information Processing Systems (eds Bengio, S. et al.) 3201\u20133211 (Curran Associates, 2018)."},{"key":"70_CR33","doi-asserted-by":"publisher","first-page":"45","DOI":"10.1023\/A:1018972901171","volume":"90","author":"A Br\u00fcngger","year":"1999","unstructured":"Br\u00fcngger, A., Marzetta, A., Fukuda, K. & Nievergelt, J. The parallel search bench ZRAM and its applications. Ann. Oper. Res. 90, 45\u201363 (1999).","journal-title":"Ann. Oper. Res."},{"key":"70_CR34","doi-asserted-by":"publisher","first-page":"26","DOI":"10.1145\/1455248.1455250","volume":"55","author":"RE Korf","year":"2008","unstructured":"Korf, R. E. Linear-time disk-based implicit graph search. JACM 55, 26 (2008).","journal-title":"JACM"},{"key":"70_CR35","first-page":"103","volume":"13","author":"AW Moore","year":"1993","unstructured":"Moore, A. W. & Atkeson, C. G. Prioritized sweeping: reinforcement learning with less data and less time. Mach. Learn. 13, 103\u2013130 (1993).","journal-title":"Mach. Learn."},{"key":"70_CR36","unstructured":"Newell, A. & Simon, H. A. GPS, a Program that Simulates Human Thought Technical Report (Rand Corporation, 1961)."},{"key":"70_CR37","doi-asserted-by":"publisher","first-page":"189","DOI":"10.1016\/0004-3702(71)90010-5","volume":"2","author":"RE Fikes","year":"1971","unstructured":"Fikes, R. E. & Nilsson, N. J. STRIPS: a new approach to the application of theorem proving to problem solving. Artif. Intell. 2, 189\u2013208 (1971).","journal-title":"Artif. Intell."},{"key":"70_CR38","unstructured":"Anthony, T., Tian, Z. & Barber, D. Thinking fast and slow with deep learning and tree search. In Advances in Neural Information Processing Systems (eds Guyon, I. et al.) 5360\u20135370 (Curran Associates, 2017)."},{"key":"70_CR39","doi-asserted-by":"crossref","unstructured":"Wilt, C. M. & Ruml, W. When does weighted A* fail? In Proc. SOCS (eds Borrajo, D. et al.) 137\u2013144 (AAAI Press, 2012).","DOI":"10.1609\/socs.v3i1.18250"},{"key":"70_CR40","unstructured":"Ioffe, S. & Szegedy, C. Batch normalization: accelerating deep network training by reducing internal covariate shift. In Proceedings of International Conference on Machine Learning (eds Bach, F. & Blei, D.) 448\u2013456 (PMLR, 2015)."},{"key":"70_CR41","unstructured":"Glorot, X., Bordes, A. & Bengio, Y. Deep sparse rectifier neural networks. In Proceedings of the Fourteenth International Conference on Artificial Intelligence and Statistics (eds Gordon, G., Dunson, D. & Dud\u00edk, M.) 315\u2013323 (PMLR, 2011)."},{"key":"70_CR42","unstructured":"Kingma, D. P. & Ba, J. Adam: a method for stochastic optimization. In Proceedings of International Conference on Learning Representations (ICLR) (eds Bach, F. & Blei, D.) (PMLR, 2015)."},{"key":"70_CR43","unstructured":"Samadi, M., Felner, A. & Schaeffer, J. Learning from multiple heuristics. In Proceedings of the 23rd National Conference on Artificial Intelligence (ed. Cohn, A.) (AAAI Press, 2008)."},{"key":"70_CR44","doi-asserted-by":"publisher","unstructured":"Agostinelli, F., McAleer, S., Shmakov, A. & Baldi, P. Learning to Solve the Rubiks Cube (Code Ocean, 2019); https:\/\/doi.org\/10.24433\/CO.4958495.v1","DOI":"10.24433\/CO.4958495.v1"}],"container-title":["Nature Machine Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.nature.com\/articles\/s42256-019-0070-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.nature.com\/articles\/s42256-019-0070-z","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.nature.com\/articles\/s42256-019-0070-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,17]],"date-time":"2022-12-17T19:40:29Z","timestamp":1671306029000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.nature.com\/articles\/s42256-019-0070-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,7,15]]},"references-count":44,"journal-issue":{"issue":"8","published-online":{"date-parts":[[2019,8]]}},"alternative-id":["70"],"URL":"https:\/\/doi.org\/10.1038\/s42256-019-0070-z","relation":{},"ISSN":["2522-5839"],"issn-type":[{"value":"2522-5839","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,7,15]]},"assertion":[{"value":"23 January 2019","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 June 2019","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 July 2019","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare no competing interests.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}]}}