{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,13]],"date-time":"2025-10-13T05:12:17Z","timestamp":1760332337668,"version":"build-2065373602"},"publisher-location":"Cham","reference-count":32,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032084613","type":"print"},{"value":"9783032084620","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,10,14]],"date-time":"2025-10-14T00:00:00Z","timestamp":1760400000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,10,14]],"date-time":"2025-10-14T00:00:00Z","timestamp":1760400000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-08462-0_24","type":"book-chapter","created":{"date-parts":[[2025,10,13]],"date-time":"2025-10-13T04:47:19Z","timestamp":1760330839000},"page":"301-312","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Action Space Size Effects in\u00a0Reinforcement Learning for\u00a0the\u00a0Vehicle Routing Problem"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-7648-0115","authenticated-orcid":false,"given":"Jon","family":"D\u00edaz-Aparicio","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4092-0245","authenticated-orcid":false,"given":"Gabriel","family":"Duflo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9355-3610","authenticated-orcid":false,"given":"Jenny","family":"Fajardo-Calderin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9581-1823","authenticated-orcid":false,"given":"Enrique","family":"Onieva","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,10,14]]},"reference":[{"key":"24_CR1","unstructured":"What is reinforcement learning? https:\/\/de.mathworks.com\/help\/reinforcement-learning\/ug\/what-is-reinforcement-learning.html. Accessed 1 May 2025"},{"issue":"7","key":"24_CR2","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3543846","volume":"55","author":"MM Afsar","year":"2022","unstructured":"Afsar, M.M., Crump, T., Far, B.: Reinforcement learning based recommender systems: a survey. ACM Comput. Surv. 55(7), 1\u201338 (2022)","journal-title":"ACM Comput. Surv."},{"key":"24_CR3","doi-asserted-by":"publisher","DOI":"10.1016\/j.tre.2021.102496","volume":"157","author":"R Basso","year":"2022","unstructured":"Basso, R., Kulcs\u00e1r, B., Sanchez-Diaz, I., Qu, X.: Dynamic stochastic electric vehicle routing with safe reinforcement learning. Trans. res. part E: C and trans. rev. 157, 102496 (2022)","journal-title":"Trans. res. part E: C and trans. rev."},{"issue":"3731","key":"24_CR4","first-page":"34","volume":"153","author":"R Bellman","year":"1966","unstructured":"Bellman, R.: Dynamic programming. science 153(3731), 34\u201337 (1966)","journal-title":"Dynamic programming. science"},{"key":"24_CR5","doi-asserted-by":"publisher","unstructured":"Correll, R., Weinberg, S., Sanches, F., Ide, T., Suzuki, T.: Reinforcement learning for multi-truck vehicle routing problems. ArXiv abs\/2211.17078, \u2013 (2022). https:\/\/doi.org\/10.48550\/arXiv.2211.17078","DOI":"10.48550\/arXiv.2211.17078"},{"key":"24_CR6","doi-asserted-by":"crossref","unstructured":"Diaz, J., Fajardo, J., Rodriguez, E., Onieva, E.: A flexible vehicle routing reinforcement learning environment for the reusability of trained agents. In: Proceedings of the 2024 10th International Conference on Computer Technology Applications, pp. 128\u2013134 (2024)","DOI":"10.1145\/3674558.3674576"},{"key":"24_CR7","unstructured":"Dulac-Arnold, G., et al.: Deep reinforcement learning in large discrete action spaces. arXiv preprint arXiv:1512.07679 (2015)"},{"key":"24_CR8","doi-asserted-by":"publisher","DOI":"10.1016\/j.jobe.2022.104165","volume":"50","author":"Q Fu","year":"2022","unstructured":"Fu, Q., Han, Z., Chen, J., Lu, Y., Wu, H., Wang, Y.: Applications of reinforcement learning for building energy efficiency control: a review. J. Build. Eng. 50, 104165 (2022)","journal-title":"J. Build. Eng."},{"key":"24_CR9","doi-asserted-by":"crossref","unstructured":"Gonzalez-Santocildes, A., Vazquez, J.I., Eguiluz, A.: Enhancing robot behavior with eeg, reinforcement learning and beyond: a review of techniques in collaborative robotics. Appl. Sci. (2076-3417) 14(14) (2024)","DOI":"10.3390\/app14146345"},{"issue":"2","key":"24_CR10","doi-asserted-by":"publisher","first-page":"895","DOI":"10.1007\/s10462-021-09996-w","volume":"55","author":"S Gronauer","year":"2022","unstructured":"Gronauer, S., Diepold, K.: Multi-agent deep reinforcement learning: a survey. Artif. Intell. Rev. 55(2), 895\u2013943 (2022)","journal-title":"Artif. Intell. Rev."},{"key":"24_CR11","doi-asserted-by":"crossref","unstructured":"Kanervisto, A., Scheller, C., Hautam\u00e4ki, V.: Action space shaping in deep reinforcement learning. In: 2020 IEEE conference on games (CoG), pp. 479\u2013486. IEEE (2020)","DOI":"10.1109\/CoG47356.2020.9231687"},{"key":"24_CR12","doi-asserted-by":"publisher","first-page":"11528","DOI":"10.1109\/TITS.2021.3105232","volume":"23","author":"B Lin","year":"2020","unstructured":"Lin, B., Ghaddar, B., Nathwani, J.: Deep reinforcement learning for the electric vehicle routing problem with time windows. IEEE Trans. Intell. Transp. Syst. 23, 11528\u201311538 (2020). https:\/\/doi.org\/10.1109\/TITS.2021.3105232","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"24_CR13","doi-asserted-by":"publisher","unstructured":"Lyu, L., Shen, Y., Zhang, S.: The advance of reinforcement learning and deep reinforcement learning. In: 2022 IEEE International Conference on Electrical Engineering, Big Data and Algorithms (EEBDA), pp. 644\u2013648 (2022).https:\/\/doi.org\/10.1109\/EEBDA53927.2022.9744760","DOI":"10.1109\/EEBDA53927.2022.9744760"},{"key":"24_CR14","doi-asserted-by":"crossref","unstructured":"Majeed, S.J., Hutter, M.: Exact reduction of huge action spaces in general reinforcement learning. In: Proceedings of the AAAI Conference on Artificial Intelligence. vol.\u00a035, pp. 8874\u20138883 (2021)","DOI":"10.1609\/aaai.v35i10.17074"},{"issue":"2","key":"24_CR15","first-page":"137","volume":"2","author":"D Michie","year":"1968","unstructured":"Michie, D., Chambers, R.A.: Boxes: an experiment in adaptive control. Machine intell. 2(2), 137\u2013152 (1968)","journal-title":"Machine intell."},{"key":"24_CR16","unstructured":"Mnih, V., et al.: Asynchronous methods for deep reinforcement learning. In: International conference on machine learning, pp. 1928\u20131937. PmLR (2016)"},{"key":"24_CR17","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2022.116830","volume":"199","author":"M Noaeen","year":"2022","unstructured":"Noaeen, M., et al.: Reinforcement learning in urban network traffic signal control: a systematic literature review. Expert Syst. Appl. 199, 116830 (2022)","journal-title":"Expert Syst. Appl."},{"issue":"1","key":"24_CR18","doi-asserted-by":"publisher","first-page":"405","DOI":"10.1007\/s10489-022-03456-w","volume":"53","author":"W Pan","year":"2023","unstructured":"Pan, W., Liu, S.Q.: Deep reinforcement learning for the dynamic and uncertain vehicle routing problem. Appl. Intell. 53(1), 405\u2013422 (2023)","journal-title":"Appl. Intell."},{"key":"24_CR19","doi-asserted-by":"publisher","unstructured":"Qin, W., Zhuang, Z., Huang, Z., Huang, H.: A novel reinforcement learning-based hyper-heuristic for heterogeneous vehicle routing problem. Comput. Ind. Eng. 156, 107252 (2021). https:\/\/doi.org\/10.1016\/J.CIE.2021.107252","DOI":"10.1016\/J.CIE.2021.107252"},{"key":"24_CR20","doi-asserted-by":"publisher","unstructured":"Raza, S.M., Sajid, M., Singh, J.: Vehicle routing problem using reinforcement learning: Recent advancements. Lect. Notes Electr. Eng. 858, 269\u2013280 (2022). https:\/\/doi.org\/10.1007\/978-981-19-0840-8_20","DOI":"10.1007\/978-981-19-0840-8_20"},{"key":"24_CR21","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., Klimov, O.: Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)"},{"key":"24_CR22","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2023.120495","volume":"231","author":"AK Shakya","year":"2023","unstructured":"Shakya, A.K., Pillai, G., Chakrabarty, S.: Reinforcement learning algorithms: a brief survey. Expert Syst. Appl. 231, 120495 (2023)","journal-title":"Expert Syst. Appl."},{"issue":"2","key":"24_CR23","doi-asserted-by":"publisher","first-page":"945","DOI":"10.1007\/s10462-021-09997-9","volume":"55","author":"B Singh","year":"2022","unstructured":"Singh, B., Kumar, R., Singh, V.P.: Reinforcement learning in robotic applications: a comprehensive survey. Artif. Intell. Rev. 55(2), 945\u2013990 (2022)","journal-title":"Artif. Intell. Rev."},{"issue":"2","key":"24_CR24","doi-asserted-by":"publisher","first-page":"254","DOI":"10.1287\/opre.35.2.254","volume":"35","author":"MM Solomon","year":"1987","unstructured":"Solomon, M.M.: Algorithms for the vehicle routing and scheduling problems with time window constraints. Oper. Res. 35(2), 254\u2013265 (1987)","journal-title":"Oper. Res."},{"key":"24_CR25","doi-asserted-by":"publisher","unstructured":"Wang, C., Cao, Z., Wu, Y., Teng, L., Wu, G.: Deep reinforcement learning for solving vehicle routing problems with backhauls. In: IEEE Transactions on Neural Networks and Learning Systems 36, 4779\u20134793 (2024). https:\/\/doi.org\/10.1109\/TNNLS.2024.3371781","DOI":"10.1109\/TNNLS.2024.3371781"},{"issue":"4","key":"24_CR26","doi-asserted-by":"publisher","first-page":"5064","DOI":"10.1109\/TNNLS.2022.3207346","volume":"35","author":"X Wang","year":"2022","unstructured":"Wang, X., et al.: Deep reinforcement learning: a survey. IEEE Trans. Neural Netw. Learn. Syst. 35(4), 5064\u20135078 (2022)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"24_CR27","first-page":"279","volume":"8","author":"CJ Watkins","year":"1992","unstructured":"Watkins, C.J., Dayan, P.: Q-learning. Mach. learn. 8, 279\u2013292 (1992)","journal-title":"Q-learning. Mach. learn."},{"key":"24_CR28","doi-asserted-by":"publisher","first-page":"11107","DOI":"10.1109\/TCYB.2021.3089179","volume":"52","author":"Y Xu","year":"2021","unstructured":"Xu, Y., Fang, M., Chen, L., Xu, G., Du, Y., Zhang, C.: Reinforcement learning with multiple relational attention for solving vehicle routing problems. IEEE Trans. Cybern. 52, 11107\u201311120 (2021). https:\/\/doi.org\/10.1109\/TCYB.2021.3089179","journal-title":"IEEE Trans. Cybern."},{"key":"24_CR29","doi-asserted-by":"publisher","first-page":"3806","DOI":"10.1109\/TITS.2019.2909109","volume":"20","author":"J Yu","year":"2019","unstructured":"Yu, J., Yu, W., Gu, J.: Online vehicle routing with neural combinatorial optimization and deep reinforcement learning. IEEE Trans. Intell. Transp. Syst. 20, 3806\u20133817 (2019). https:\/\/doi.org\/10.1109\/TITS.2019.2909109","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"24_CR30","doi-asserted-by":"publisher","unstructured":"Zhang, K., He, F., Zhang, Z., Lin, X., Li, M.: Multi-vehicle routing problems with soft time windows: A multi-agent reinforcement learning approach. Trans. Res. Part C: Emerg. Technol. 121, 102861 (2020). https:\/\/doi.org\/10.1016\/J.TRC.2020.102861","DOI":"10.1016\/J.TRC.2020.102861"},{"key":"24_CR31","doi-asserted-by":"publisher","first-page":"7208","DOI":"10.1109\/tits.2020.3003163","volume":"22","author":"J Zhao","year":"2021","unstructured":"Zhao, J., Mao, M., Zhao, X., Zou, J.: A hybrid of deep reinforcement learning and local search for the vehicle routing problems. IEEE Trans. Intell. Transp. Syst. 22, 7208\u20137218 (2021). https:\/\/doi.org\/10.1109\/tits.2020.3003163","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"issue":"2","key":"24_CR32","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3625232","volume":"15","author":"Z Zong","year":"2024","unstructured":"Zong, Z., Tong, X., Zheng, M., Li, Y.: Reinforcement learning for solving multiple vehicle routing problem with time window. ACM Trans. Intell. Syst. Technol. 15(2), 1\u201319 (2024)","journal-title":"ACM Trans. Intell. Syst. Technol."}],"container-title":["Lecture Notes in Computer Science","Hybrid Artificial Intelligent Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-08462-0_24","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,13]],"date-time":"2025-10-13T04:47:23Z","timestamp":1760330843000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-08462-0_24"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,14]]},"ISBN":["9783032084613","9783032084620"],"references-count":32,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-08462-0_24","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,10,14]]},"assertion":[{"value":"14 October 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"HAIS","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Hybrid Artificial Intelligence Systems","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Salamanca","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Spain","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16 October 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17 October 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"hais2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/haisconference.eu","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}