{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T15:16:46Z","timestamp":1783091806851,"version":"3.54.6"},"reference-count":43,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["72372088"],"award-info":[{"award-number":["72372088"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100017610","name":"Shenzhen Science and Technology Innovation Program","doi-asserted-by":"publisher","award":["GJHZ20220913143003006"],"award-info":[{"award-number":["GJHZ20220913143003006"]}],"id":[{"id":"10.13039\/501100017610","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100017610","name":"Shenzhen Science and Technology Innovation Program","doi-asserted-by":"publisher","award":["JCYJ20240813112031041"],"award-info":[{"award-number":["JCYJ20240813112031041"]}],"id":[{"id":"10.13039\/501100017610","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.eswa.2026.132847","type":"journal-article","created":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T23:51:21Z","timestamp":1779234681000},"page":"132847","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Reinforcement Learning for Pod Retrieval Scheduling and Robot Dispatching in the Multi-deep Compact Robotic Mobile Fulfillment Systems"],"prefix":"10.1016","volume":"327","author":[{"given":"Jie","family":"Shao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mingzhe","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Binjia","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3678-6522","authenticated-orcid":false,"given":"Mingyao","family":"Qi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6780-2847","authenticated-orcid":false,"given":"Peng","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"5","key":"10.1016\/j.eswa.2026.132847_bib0001","doi-asserted-by":"crossref","first-page":"1695","DOI":"10.1007\/s10845-010-0471-7","article-title":"Heuristics for puzzle-based storage systems driven by a limited set of automated guided vehicles","volume":"23","author":"Alfieri","year":"2012","journal-title":"Journal of Intelligent Manufacturing"},{"key":"10.1016\/j.eswa.2026.132847_bib0002","unstructured":"Bello, I., Pham, H., Le, Q., Norouzi, M., & Bengio, S. (2016). Neural combinatorial optimization with reinforcement learningarXiv: 1611.09940."},{"issue":"2","key":"10.1016\/j.eswa.2026.132847_bib0003","doi-asserted-by":"crossref","first-page":"550","DOI":"10.1016\/j.ejor.2017.03.053","article-title":"Parts-to-picker based order processing in a rack-moving mobile robots environment","volume":"262","author":"Boysen","year":"2017","journal-title":"European Journal of Operational Research"},{"key":"10.1016\/j.eswa.2026.132847_bib0004","doi-asserted-by":"crossref","first-page":"348","DOI":"10.1016\/j.trb.2022.11.002","article-title":"A comprehensive toolbox for load retrieval in puzzle-based storage systems with simultaneous movements","volume":"166","author":"Bukchin","year":"2022","journal-title":"Transportation Research Part B: Methodological"},{"issue":"2","key":"10.1016\/j.eswa.2026.132847_bib0005","doi-asserted-by":"crossref","first-page":"825","DOI":"10.1109\/TASE.2018.2862380","article-title":"Scheduling semiconductor testing facility by using cuckoo search algorithm with reinforcement learning and surrogate modeling","volume":"16","author":"Cao","year":"2018","journal-title":"IEEE Transactions on Automation Science and Engineering"},{"key":"10.1016\/j.eswa.2026.132847_bib0006","doi-asserted-by":"crossref","DOI":"10.1016\/j.tre.2020.102087","article-title":"Robot scheduling for pod retrieval in a robotic mobile fulfillment system","volume":"142","author":"Gharehgozli","year":"2020","journal-title":"Transportation Research Part E"},{"issue":"16","key":"10.1016\/j.eswa.2026.132847_bib0007","doi-asserted-by":"crossref","first-page":"5032","DOI":"10.1080\/00207543.2020.1779370","article-title":"Robotic mobile fulfillment systems considering customer classes","volume":"59","author":"Gong","year":"2021","journal-title":"International Journal of Production Research"},{"issue":"2","key":"10.1016\/j.eswa.2026.132847_bib0008","doi-asserted-by":"crossref","first-page":"429","DOI":"10.1109\/TASE.2013.2278252","article-title":"Gridstore: A puzzle-based storage system with decentralized control","volume":"11","author":"Gue","year":"2013","journal-title":"IEEE Transactions on Automation Science and Engineering"},{"issue":"2","key":"10.1016\/j.eswa.2026.132847_bib0009","doi-asserted-by":"crossref","first-page":"820","DOI":"10.1016\/j.ejor.2022.03.042","article-title":"Reinforcement learning for multi-item retrieval in the puzzle-based storage system","volume":"305","author":"He","year":"2023","journal-title":"European Journal of Operational Research"},{"issue":"3","key":"10.1016\/j.eswa.2026.132847_bib0010","doi-asserted-by":"crossref","first-page":"141","DOI":"10.1007\/BF00339943","article-title":"\u2018Neural\u2019 computation of decisions in optimization problems","volume":"52","author":"Hopfield","year":"1985","journal-title":"Biological Cybernetics"},{"key":"10.1016\/j.eswa.2026.132847_bib0011","series-title":"Proc. IEEE 7th int. conf. ind. eng. appl. (ICIEA)","first-page":"1","article-title":"Model-based optimization of pod point matching decision in robotic mobile fulfillment system","author":"Ji","year":"2020"},{"key":"10.1016\/j.eswa.2026.132847_bib0012","unstructured":"Kingma, D., & Ba, J. (2014). Adam: a method for stochastic optimization. arXiv preprint arXiv: 1412.6980."},{"issue":"3","key":"10.1016\/j.eswa.2026.132847_bib0013","doi-asserted-by":"crossref","first-page":"976","DOI":"10.1016\/j.ejor.2016.06.063","article-title":"Estimating performance in a robotic mobile fulfillment system","volume":"256","author":"Lamballais","year":"2017","journal-title":"European Journal of Operational Research"},{"key":"10.1016\/j.eswa.2026.132847_bib0014","doi-asserted-by":"crossref","DOI":"10.1016\/j.simpat.2021.102366","article-title":"A simulation study on the robotic mobile fulfillment system in high-density storage warehouses","volume":"112","author":"Li","year":"2021","journal-title":"Simulation Modelling Practice and Theory"},{"issue":"1","key":"10.1016\/j.eswa.2026.132847_bib0015","doi-asserted-by":"crossref","first-page":"51","DOI":"10.1016\/j.ejor.2021.09.032","article-title":"An efficient heuristic for minimizing the number of moves for the retrieval of a single item in a puzzle-based storage system with multiple escorts","volume":"301","author":"Ma","year":"2022","journal-title":"European Journal of Operational Research"},{"key":"10.1016\/j.eswa.2026.132847_bib0016","doi-asserted-by":"crossref","DOI":"10.1016\/j.orp.2019.100128","article-title":"Decision rules for robotic mobile fulfillment systems","volume":"6","author":"Merschformann","year":"2019","journal-title":"Operations Research Perspectives"},{"issue":"21","key":"10.1016\/j.eswa.2026.132847_bib0017","doi-asserted-by":"crossref","first-page":"6423","DOI":"10.1080\/00207543.2017.1304660","article-title":"Modelling load retrievals in puzzle-based storage systems","volume":"55","author":"Mirzaei","year":"2017","journal-title":"International Journal of Production Research"},{"key":"10.1016\/j.eswa.2026.132847_bib0018","unstructured":"Mnih, V., Kavukcuoglu, K., Silver, D., Graves, A., Antonoglou, I., Wierstra, D., & Riedmiller, M. (2013). Playing atari with deep reinforcement learning. arXiv preprint arXiv: 1312.5602."},{"issue":"7540","key":"10.1016\/j.eswa.2026.132847_bib0019","doi-asserted-by":"crossref","first-page":"529","DOI":"10.1038\/nature14236","article-title":"Human-level control through deep reinforcement learning","volume":"518","author":"Mnih","year":"2015","journal-title":"Nature"},{"key":"10.1016\/j.eswa.2026.132847_bib0020","series-title":"Proc. winter simulation conference (wsc)","first-page":"1","article-title":"Comparison of deadlock handling strategies for different warehouse layouts with an agvs","author":"M\u00fcller","year":"2020"},{"key":"10.1016\/j.eswa.2026.132847_bib0021","series-title":"Advances in neural information processing systems","article-title":"Reinforcement learning for solving the vehicle routing problem","volume":"vol. 31","author":"Nazari","year":"2018"},{"key":"10.1016\/j.eswa.2026.132847_bib0022","series-title":"Proc. IIE annual conference","first-page":"1","article-title":"Retrieval time performance in puzzle-based storage systems","author":"Rohit","year":"2010"},{"key":"10.1016\/j.eswa.2026.132847_bib0023","doi-asserted-by":"crossref","first-page":"119","DOI":"10.1016\/j.tre.2018.11.005","article-title":"Robot-storage zone assignment strategies in mobile fulfillment systems","volume":"122","author":"Roy","year":"2019","journal-title":"Transportation Research Part E: Logistics and Transportation Review"},{"key":"10.1016\/j.eswa.2026.132847_bib0024","unstructured":"Schaul, T., Quan, J., Antonoglou, I., & Silver, D. (2015). Prioritized experience replay,. arXiv: 1511.05952."},{"issue":"7676","key":"10.1016\/j.eswa.2026.132847_bib0025","doi-asserted-by":"crossref","first-page":"354","DOI":"10.1038\/nature24270","article-title":"Mastering the game of go without human knowledge","volume":"550","author":"Silver","year":"2017","journal-title":"Nature"},{"issue":"1","key":"10.1016\/j.eswa.2026.132847_bib0026","doi-asserted-by":"crossref","first-page":"15","DOI":"10.1287\/ijoc.11.1.15","article-title":"Neural networks for combinatorial optimization: a review of more than a decade of research","volume":"11","author":"Smith","year":"1999","journal-title":"INFORMS Journal on Computing"},{"key":"10.1016\/j.eswa.2026.132847_bib0027","series-title":"Proc. AAAI Conf. Artif.","article-title":"Deep reinforcement learning with double q-learning","author":"Van Hasselt","year":"2016"},{"key":"10.1016\/j.eswa.2026.132847_bib0028","series-title":"Advances in neural information processing systems 30","article-title":"Attention is all you need","author":"Vaswani","year":"2017"},{"key":"10.1016\/j.eswa.2026.132847_bib0029","series-title":"Advances in neural information processing systems","article-title":"Pointer networks","volume":"vol. 28","author":"Vinyals","year":"2015"},{"issue":"4","key":"10.1016\/j.eswa.2026.132847_bib0030","doi-asserted-by":"crossref","first-page":"1995","DOI":"10.1109\/TMECH.2016.2547959","article-title":"Three-dimensional cell rotation with fluidic flow-controlled cell manipulating device","volume":"21","author":"Wang","year":"2016","journal-title":"IEEE\/ASME Transactions on Mechatronics"},{"issue":"6","key":"10.1016\/j.eswa.2026.132847_bib0031","doi-asserted-by":"crossref","first-page":"1479","DOI":"10.1287\/trsc.2018.0826","article-title":"Storage assignment with rack-moving mobile robots in KIVA warehouses","volume":"52","author":"Weidinger","year":"2018","journal-title":"Transportation Science"},{"issue":"1","key":"10.1016\/j.eswa.2026.132847_bib0032","first-page":"9","article-title":"Coordinating hundreds of cooperative, autonomous vehicles in warehouses","volume":"29","author":"Wurman","year":"2008","journal-title":"AI Magazine"},{"issue":"1","key":"10.1016\/j.eswa.2026.132847_bib0033","doi-asserted-by":"crossref","first-page":"143","DOI":"10.1080\/00207543.2018.1461952","article-title":"An optimal and a heuristic algorithm for the single-item retrieval problem in puzzle-based storage systems with multiple escorts","volume":"57","author":"Yalcin","year":"2019","journal-title":"International Journal of Production Research"},{"issue":"15","key":"10.1016\/j.eswa.2026.132847_bib0034","doi-asserted-by":"crossref","first-page":"4727","DOI":"10.1080\/00207543.2021.1936264","article-title":"Modelling and analysis for multi-deep compact robotic mobile fulfilment system","volume":"60","author":"Yang","year":"2022","journal-title":"International Journal of Production Research"},{"issue":"2","key":"10.1016\/j.eswa.2026.132847_bib0035","doi-asserted-by":"crossref","first-page":"156","DOI":"10.1080\/24725854.2021.2010151","article-title":"Dense and fast: Achieving shortest unimpeded retrieval with a minimum number of empty cells in puzzle-based storage systems","volume":"55","author":"Yu","year":"2022","journal-title":"IISE Transactions"},{"issue":"1","key":"10.1016\/j.eswa.2026.132847_bib0036","doi-asserted-by":"crossref","first-page":"83","DOI":"10.1109\/TEM.2016.2634540","article-title":"Bot-in-time delivery for robotic mobile fulfillment systems","volume":"64","author":"Yuan","year":"2017","journal-title":"IEEE Transactions on Engineering Management"},{"issue":"11","key":"10.1016\/j.eswa.2026.132847_bib0037","doi-asserted-by":"crossref","first-page":"7922","DOI":"10.1109\/TSMC.2025.3598298","article-title":"Learning-based approach to integrated operational optimization problems in robot-assisted multistation warehouse systems","volume":"55","author":"Zhao","year":"2025","journal-title":"IEEE Transactions on Systems, Man, and Cybernetics: Systems"},{"issue":"9","key":"10.1016\/j.eswa.2026.132847_bib0038","doi-asserted-by":"crossref","first-page":"16314","DOI":"10.1109\/JIOT.2024.3352658","article-title":"Order picking optimization in smart warehouses with human-robot collaboration","volume":"11","author":"Zhao","year":"2024","journal-title":"IEEE Internet of Things Journal"},{"key":"10.1016\/j.eswa.2026.132847_bib0039","doi-asserted-by":"crossref","first-page":"6223","DOI":"10.1109\/TASE.2024.3440169","article-title":"Lexicographic dual-objective path finding in multi-agent systems","volume":"22","author":"Zhao","year":"2025","journal-title":"IEEE Transactions on Automation Science and Engineering"},{"issue":"2","key":"10.1016\/j.eswa.2026.132847_bib0040","doi-asserted-by":"crossref","first-page":"527","DOI":"10.1016\/j.ejor.2021.08.003","article-title":"Order picking optimization with rack-moving mobile robots and multiple workstations","volume":"300","author":"Zhuang","year":"2022","journal-title":"European Journal of Operational Research"},{"issue":"2","key":"10.1016\/j.eswa.2026.132847_bib0041","doi-asserted-by":"crossref","first-page":"733","DOI":"10.1016\/j.ejor.2017.12.008","article-title":"Evaluating battery charging and swapping strategies in a robotic mobile fulfillment system","volume":"267","author":"Zou","year":"2018","journal-title":"European Journal of Operational Research"},{"key":"10.1016\/j.eswa.2026.132847_bib0042","unstructured":"Zou, Y., & Qi, M. (2021). A heuristic method for load retrievals route programming in puzzle-based storage systems. arXiv preprint arXiv: 2102.09274."},{"key":"10.1016\/j.eswa.2026.132847_bib0043","article-title":"A building-block-based genetic algo rithm for solving the robots allocation problem in a robotic mobile fulfil ment system","author":"Zhang","year":"2019","journal-title":"Mathematical Problems in Engineering"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426017604?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426017604?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T14:22:36Z","timestamp":1783088556000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426017604"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":43,"alternative-id":["S0957417426017604"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132847","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Reinforcement Learning for Pod Retrieval Scheduling and Robot Dispatching in the Multi-deep Compact Robotic Mobile Fulfillment Systems","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132847","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"132847"}}