{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T23:05:10Z","timestamp":1779318310526,"version":"3.51.4"},"publisher-location":"Cham","reference-count":32,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032214768","type":"print"},{"value":"9783032214775","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-21477-5_24","type":"book-chapter","created":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T22:06:47Z","timestamp":1779314807000},"page":"350-364","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Incremental Learning and\u00a0Reward Shaping Strategies for\u00a0Deep Reinforcement Learning upon\u00a0CVRP"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7497-2756","authenticated-orcid":false,"given":"Mert","family":"Alag\u00f6zl\u00fc","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,5,1]]},"reference":[{"key":"24_CR1","unstructured":"Ardon, L.: Reinforcement learning to solve NP-hard problems: An application to the CVRP. arXiv preprint abs\/2201.05393 (2022). https:\/\/arxiv.org\/abs\/2201.05393"},{"key":"24_CR2","doi-asserted-by":"publisher","unstructured":"Bengio, Y., Louradour, J., Collobert, R., Weston, J.: Curriculum learning. In: Proceedings of the 26th International Conference on Machine Learning (ICML 2009) (2009). https:\/\/doi.org\/10.1145\/1553374.1553380","DOI":"10.1145\/1553374.1553380"},{"key":"24_CR3","doi-asserted-by":"publisher","unstructured":"Bjorck, J., Gomes, C.P., Weinberger, K.Q.: Is high variance unavoidable in RL? a case study in continuous control. arXiv preprint (2021). https:\/\/doi.org\/10.48550\/arXiv.2110.11222, https:\/\/arxiv.org\/abs\/2110.11222","DOI":"10.48550\/arXiv.2110.11222"},{"key":"24_CR4","doi-asserted-by":"crossref","unstructured":"Cappart, Q., Moisan, T., Rousseau, L., Pr\u00e9mont-Schwarz, I., Cir\u00e9, A.A.: Combining reinforcement learning and constraint programming for combinatorial optimization. In: Proceedings of the AAAI Conference on Artificial Intelligence. vol.\u00a035, pp. 3677\u20133687. AAAI Press (2021)","DOI":"10.1609\/aaai.v35i5.16484"},{"issue":"1","key":"24_CR5","doi-asserted-by":"publisher","first-page":"80","DOI":"10.1287\/mnsc.6.1.80","volume":"6","author":"GB Dantzig","year":"1959","unstructured":"Dantzig, G.B., Ramser, J.H.: The truck dispatching problem. Manage. Sci. 6(1), 80\u201391 (1959). https:\/\/doi.org\/10.1287\/mnsc.6.1.80","journal-title":"Manage. Sci."},{"key":"24_CR6","unstructured":"Delarue, A., Anderson, R., Tjandraatmadja, C.: Reinforcement learning with combinatorial actions: an application to vehicle routing (2020). https:\/\/www.semanticscholar.org\/paper\/Reinforcement-Learning-with-Combinatorial-Actions%3A-Delarue-Anderson\/c7f0bac65336214c042920a378b7ecd511a10ce5"},{"key":"24_CR7","doi-asserted-by":"publisher","DOI":"10.1016\/j.jclepro.2020.125444","volume":"285","author":"H D\u00fcndar","year":"2020","unstructured":"D\u00fcndar, H., \u00d6m\u00fcrg\u00f6n\u00fcl\u015fen, M., Soysal, M.: A review on sustainable urban vehicle routing. J. Clean. Prod. 285, 125444 (2020). https:\/\/doi.org\/10.1016\/j.jclepro.2020.125444","journal-title":"J. Clean. Prod."},{"key":"24_CR8","doi-asserted-by":"publisher","unstructured":"Eschmann, J.: Reward function design in reinforcement learning. In: Reinforcement Learning Algorithms: Analysis and Applications, pp. 25\u201333. Springer, Cham (2021). https:\/\/doi.org\/10.1007\/978-3-030-41188-6_3","DOI":"10.1007\/978-3-030-41188-6_3"},{"key":"24_CR9","doi-asserted-by":"publisher","unstructured":"Golden, B., Wang, X., Wasil, E.: The evolution of the vehicle routing problem\u2014a survey of VRP research and practice from 2005 to 2022. In: The Evolution of the Vehicle Routing Problem: A Survey of VRP Research and Practice from 2005 to 2022, pp. 1\u201364. Springer, Cham (2023). https:\/\/doi.org\/10.1007\/978-3-031-18716-2_1","DOI":"10.1007\/978-3-031-18716-2_1"},{"key":"24_CR10","unstructured":"Gupta, D., Chandak, Y., Jordan, S.M., Thomas, P.S., da\u00a0Silva, B.C.: Behavior alignment via reward function optimization. In: Proceedings of the 37th Conference on Neural Information Processing Systems (NeurIPS 2023). NeurIPS \u201923, Curran Associates Inc., New Orleans, LA, USA (2023)"},{"issue":"12","key":"24_CR11","doi-asserted-by":"publisher","first-page":"1028","DOI":"10.1016\/j.tics.2020.09.004","volume":"24","author":"R Hadsell","year":"2020","unstructured":"Hadsell, R., Rao, D., Rusu, A.A., Pascanu, R.: Embracing change: Continual learning in deep neural networks. Trends Cogn. Sci. 24(12), 1028\u20131040 (2020). https:\/\/doi.org\/10.1016\/j.tics.2020.09.004","journal-title":"Trends Cogn. Sci."},{"issue":"4","key":"24_CR12","doi-asserted-by":"publisher","first-page":"345","DOI":"10.7307\/ptt.v27i4.1616","volume":"27","author":"H Han","year":"2015","unstructured":"Han, H., Ponce, E.C.: Waste collection vehicle routing problem: literature review. PROMET - Traffic Transp. 27(4), 345\u2013358 (2015). https:\/\/doi.org\/10.7307\/ptt.v27i4.1616","journal-title":"PROMET - Traffic Transp."},{"issue":"10","key":"24_CR13","doi-asserted-by":"publisher","first-page":"3806","DOI":"10.1109\/TITS.2019.2909109","volume":"20","author":"J James","year":"2019","unstructured":"James, J., Yu, W., Gu, J.: Online vehicle routing with neural combinatorial optimization and deep reinforcement learning. IEEE Trans. Intell. Transp. Syst. 20(10), 3806\u20133817 (2019)","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"issue":"4","key":"24_CR14","doi-asserted-by":"publisher","first-page":"96","DOI":"10.3390\/logistics8040096","volume":"8","author":"A Konovalenko","year":"2024","unstructured":"Konovalenko, A., Hvattum, L.M.: Optimizing a dynamic vehicle routing problem with deep reinforcement learning: analyzing state-space components. Logistics 8(4), 96 (2024). https:\/\/doi.org\/10.3390\/logistics8040096","journal-title":"Logistics"},{"key":"24_CR15","doi-asserted-by":"publisher","unstructured":"Kroese, D.P., Taimre, T., Botev, Z.I.: Handbook of Monte Carlo Methods. Wiley Series in Probability and Statistics, Wiley (2011). https:\/\/doi.org\/10.1002\/9781118014967","DOI":"10.1002\/9781118014967"},{"key":"24_CR16","unstructured":"Laud, A.D.: Theory and application of reward shaping in reinforcement learning (2004). https:\/\/www.ideals.illinois.edu\/items\/10802"},{"key":"24_CR17","doi-asserted-by":"publisher","unstructured":"Li, J., et al.: Deep reinforcement learning for solving the heterogeneous capacitated vehicle routing problem. IEEE Trans. Cybern. 52(12), 13572\u201313585 (2021). https:\/\/doi.org\/10.1109\/TCYB.2021.3111082","DOI":"10.1109\/TCYB.2021.3111082"},{"key":"24_CR18","doi-asserted-by":"publisher","unstructured":"Li, Y., Wu, W., Luo, X., Zheng, M., Zhang, Y., Peng, B.: A survey: navigating the landscape of incremental learning techniques and trends. In: Proceedings of the 18th International Conference on Intelligent Systems and Knowledge Engineering (ISKE), pp. 163\u2013169. IEEE (2023). https:\/\/doi.org\/10.1109\/iske60036.2023.10481497","DOI":"10.1109\/iske60036.2023.10481497"},{"issue":"8","key":"24_CR19","doi-asserted-by":"publisher","first-page":"11528","DOI":"10.1109\/TITS.2021.3105232","volume":"23","author":"B Lin","year":"2022","unstructured":"Lin, B., Ghaddar, B., Nathwani, J.: Deep reinforcement learning for electric vehicle routing problem with time windows. IEEE Trans. Intell. Transp. Syst. 23(8), 11528\u201311538 (2022). https:\/\/doi.org\/10.1109\/TITS.2021.3105232","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"24_CR20","doi-asserted-by":"publisher","unstructured":"Lodes, L., Schiendorfer, A.: Certainty groups: A practical approach to distinguish confidence levels in neural networks. PHM Soc. Eur. Conf. 7(1), 294\u2013305 (2022). https:\/\/doi.org\/10.36001\/phme.2022.v7i1.3331","DOI":"10.36001\/phme.2022.v7i1.3331"},{"key":"24_CR21","doi-asserted-by":"publisher","DOI":"10.1016\/j.cor.2021.105400","volume":"134","author":"N Mazyavkina","year":"2021","unstructured":"Mazyavkina, N., Sviridov, S., Ivanov, S., Burnaev, E.: Reinforcement learning for combinatorial optimization: a survey. Comput. Oper. Res. 134, 105400 (2021)","journal-title":"Comput. Oper. Res."},{"key":"24_CR22","unstructured":"Narvekar, S., Peng, B., Leonetti, M., Sinapov, J., Taylor, M.E., Stone, P.: Curriculum learning for reinforcement learning domains: a framework and survey. J. Mach. Learn. Res. 21(1) (2020)"},{"key":"24_CR23","doi-asserted-by":"publisher","unstructured":"Nazari, M., Oroojlooy, A., Snyder, L.V., Tak\u00e1\u010d, M.: Reinforcement learning for solving the vehicle routing problem (2018). https:\/\/doi.org\/10.48550\/arXiv.1802.04240, http:\/\/arxiv.org\/abs\/1802.04240","DOI":"10.48550\/arXiv.1802.04240"},{"key":"24_CR24","doi-asserted-by":"publisher","unstructured":"Pendyala, A., Atamna, A., Glasmachers, T.: Solving a real-world optimization problem using proximal policy optimization with curriculum learning and reward engineering. In: Bifet, A., Krilavi\u010dius, T., Miliou, I., Nowaczyk, S. (eds.) Machine Learning and Knowledge Discovery in Databases. Applied Data Science Track. pp. 150\u2013165. Springer, Cham (2024). https:\/\/doi.org\/10.1007\/978-3-031-70381-2_10","DOI":"10.1007\/978-3-031-70381-2_10"},{"key":"24_CR25","unstructured":"Raffin, A., Hill, A., Gleave, A., Kanervisto, A., Ernestus, M., Dormann, N.: Stable-baselines3: Reliable reinforcement learning implementations. J. Mach. Learn. Res. 22(268), 1\u20138 (2021). http:\/\/jmlr.org\/papers\/v22\/20-1364.html"},{"key":"24_CR26","doi-asserted-by":"publisher","unstructured":"Shah, A., Tran, D., Tang, Y.: Efficient mitigation of bus bunching through setter-based curriculum learning. arXiv preprint (2024). https:\/\/doi.org\/10.48550\/arXiv.2405.15824, https:\/\/arxiv.org\/abs\/2405.15824","DOI":"10.48550\/arXiv.2405.15824"},{"key":"24_CR27","volume-title":"Reinforcement Learning: An Introduction","author":"RS Sutton","year":"2018","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement Learning: An Introduction, 2nd edn. MIT Press, Cambridge, MA (2018)","edition":"2"},{"key":"24_CR28","unstructured":"Taylor, M., Stone, P.: Transfer learning for reinforcement learning domains: a survey. J. Mach. Learn Res. 10, 1633\u20131685 (2009). https:\/\/www.jmlr.org\/papers\/volume10\/taylor09a\/taylor09a.pdf"},{"key":"24_CR29","doi-asserted-by":"publisher","unstructured":"Toth, P., Vigo, D.: An overview of vehicle routing problems. In: The Vehicle Routing Problem, pp. 1\u201326 (2002). https:\/\/doi.org\/10.1137\/1.9780898718515.ch1, https:\/\/epubs.siam.org\/doi\/abs\/10.1137\/1.9780898718515.ch1","DOI":"10.1137\/1.9780898718515.ch1"},{"key":"24_CR30","doi-asserted-by":"publisher","unstructured":"Voudouris, C., Tsang, E.P.K.: Guided local search. In: Handbook of Metaheuristics, vol.\u00a057, pp. 185\u2013218. Kluwer Academic Publishers, Boston (2003). https:\/\/doi.org\/10.1007\/0-306-48056-5_7","DOI":"10.1007\/0-306-48056-5_7"},{"key":"24_CR31","doi-asserted-by":"publisher","unstructured":"Wiwatcharakoses, C., Berrar, D.: Self-organizing incremental neural networks for continual learning. In: Proceedings of the Twenty-Eighth International Joint Conference on Artificial Intelligence (IJCAI-19), pp. 6476\u20136477 (2019). https:\/\/doi.org\/10.24963\/ijcai.2019\/927","DOI":"10.24963\/ijcai.2019\/927"},{"key":"24_CR32","doi-asserted-by":"publisher","DOI":"10.1016\/j.ymssp.2023.110698","author":"X Zhang","year":"2023","unstructured":"Zhang, X., Xiong, G., Ai, Y., Liu, K., Chen, L.: Vehicle dynamic dispatching using curriculum-driven reinforcement learning. Mech. Syst. Signal Process. (2023). https:\/\/doi.org\/10.1016\/j.ymssp.2023.110698","journal-title":"Mech. Syst. Signal Process."}],"container-title":["Lecture Notes in Computer Science","Machine Learning, Optimization, and Data Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-21477-5_24","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T22:06:50Z","timestamp":1779314810000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-21477-5_24"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032214768","9783032214775"],"references-count":32,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-21477-5_24","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"1 May 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"LOD","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Artificial Intelligence Symposium","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Castiglione della Pescaia","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"21 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"24 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"11","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"mod2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/lod2025.icas.events","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}