{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,24]],"date-time":"2026-02-24T06:27:46Z","timestamp":1771914466911,"version":"3.50.1"},"reference-count":35,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,11,25]],"date-time":"2025-11-25T00:00:00Z","timestamp":1764028800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"},{"start":{"date-parts":[[2025,11,25]],"date-time":"2025-11-25T00:00:00Z","timestamp":1764028800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"}],"funder":[{"DOI":"10.13039\/501100002835","name":"Chalmers University of Technology","doi-asserted-by":"crossref","id":[{"id":"10.13039\/501100002835","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Artif Intell Rev"],"abstract":"<jats:title>Abstract<\/jats:title>\n                  <jats:p>\n                    We develop a\n                    <jats:italic>deep reinforcement learning<\/jats:italic>\n                    framework for tactical decision making in an autonomous truck, specifically for Adaptive Cruise Control (ACC) and lane change maneuvers in a highway scenario. Our results demonstrate that it is beneficial to separate high-level decision-making processes and low-level control actions between the reinforcement learning agent and the low-level controllers based on physical models. In the following, we study optimizing the performance with a realistic and multi-objective reward function based on Total Cost of Operation (TCOP) of the truck using different approaches; by adding weights to reward components, by normalizing the reward components and by using curriculum learning techniques.\n                  <\/jats:p>","DOI":"10.1007\/s10462-025-11448-8","type":"journal-article","created":{"date-parts":[[2025,11,25]],"date-time":"2025-11-25T13:28:34Z","timestamp":1764077314000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Tactical decision making for autonomous trucks by deep reinforcement learning with total cost of operation based reward"],"prefix":"10.1007","volume":"59","author":[{"given":"Deepthi","family":"Pathare","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Leo","family":"Laine","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Morteza Haghir","family":"Chehreghani","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,11,25]]},"reference":[{"issue":"10","key":"11448_CR1","doi-asserted-by":"publisher","first-page":"19817","DOI":"10.1109\/TITS.2022.3160673","volume":"23","author":"L Anzalone","year":"2022","unstructured":"Anzalone L, Barra P, Barra S, Castiglione A, Nappi M (2022) An end-to-end curriculum learning approach for autonomous driving scenarios. IEEE Trans Intell Transp Syst 23(10):19817\u201319826. https:\/\/doi.org\/10.1109\/TITS.2022.3160673","journal-title":"IEEE Trans Intell Transp Syst"},{"key":"11448_CR2","doi-asserted-by":"crossref","unstructured":"Bengio Y, Louradour J, Collobert R, Weston J (2009) Curriculum learning. In: Proceedings of the 26th annual international conference on machine learning, pp 41\u201348","DOI":"10.1145\/1553374.1553380"},{"key":"11448_CR3","unstructured":"Das LC, Won M (2021) Saint-acc: safety-aware intelligent adaptive cruise control for autonomous vehicles using deep reinforcement learning. In: ICML"},{"issue":"1","key":"11448_CR4","doi-asserted-by":"publisher","first-page":"103","DOI":"10.1080\/0144164032000080494","volume":"24","author":"G De Jong","year":"2004","unstructured":"De Jong G, Gunn H, Walker W (2004) National and international freight transport models: an overview and ideas for future development. Transp Rev 24(1):103\u2013124","journal-title":"Transp Rev"},{"issue":"4","key":"11448_CR5","doi-asserted-by":"publisher","first-page":"1248","DOI":"10.1109\/TITS.2011.2157145","volume":"12","author":"C Desjardins","year":"2011","unstructured":"Desjardins C, Chaib-draa B (2011) Cooperative adaptive cruise control: a reinforcement learning approach. IEEE Trans Intell Transp Syst 12(4):1248\u20131260","journal-title":"IEEE Trans Intell Transp Syst"},{"key":"11448_CR6","unstructured":"Erdmann J (2014) Lane-changing model in sumo. In: Proceedings of the SUMO2014 modeling mobility with open data. Reports of the DLR-institute of transportation systems proceedings,"},{"issue":"3","key":"11448_CR7","doi-asserted-by":"publisher","first-page":"336","DOI":"10.3846\/transport.2010.41","volume":"25","author":"A Grislis","year":"2010","unstructured":"Grislis A (2010) Longer combination vehicles and road safety. Transport 25(3):336\u2013343","journal-title":"Transport"},{"key":"11448_CR8","unstructured":"Haarnoja T, Zhou A, Abbeel P, Levine S (2018) Soft actor-critic: off-policy maximum entropy deep reinforcement learning with a stochastic actor"},{"key":"11448_CR9","doi-asserted-by":"crossref","unstructured":"Hoel CJ, Wolff K, Laine L (2020) Tactical decision-making in autonomous driving by reinforcement learning with uncertainty estimation. In: Intelligent vehicles symposium. IEEE","DOI":"10.1109\/IV47402.2020.9304614"},{"key":"11448_CR10","doi-asserted-by":"crossref","unstructured":"Jim\u00e9nez F, Naranjo JE, Anaya JJ, Garc\u00eda F, Ponz A, Armingol JM (2016) Advanced driver assistance system for road environments to improve safety and efficiency. Transp Res Procedia 14","DOI":"10.1016\/j.trpro.2016.05.240"},{"issue":"6","key":"11448_CR11","doi-asserted-by":"publisher","first-page":"4909","DOI":"10.1109\/TITS.2021.3054625","volume":"23","author":"BR Kiran","year":"2021","unstructured":"Kiran BR, Sobh I, Talpaert V, Mannion P, Al Sallab AA, Yogamani S, P\u00e9rez P (2021) Deep reinforcement learning for autonomous driving: a survey. Trans Intell Transp Syst 23(6):4909\u20134926","journal-title":"Trans Intell Transp Syst"},{"key":"11448_CR12","unstructured":"Konda V, Tsitsiklis J (1999) Actor-critic algorithms. In: Solla, S., Leen, T., M\u00fcller, K. (eds.) Advances in neural information processing systems"},{"issue":"5","key":"11448_CR13","doi-asserted-by":"publisher","first-page":"5597","DOI":"10.1103\/PhysRevE.55.5597","volume":"55","author":"S Krauss","year":"1997","unstructured":"Krauss S, Wagner P, Gawron C (1997) Metastable states in a microscopic model of traffic flow. Phys Rev E 55(5):5597","journal-title":"Phys Rev E"},{"issue":"2","key":"11448_CR14","doi-asserted-by":"publisher","first-page":"221","DOI":"10.1109\/TIV.2020.3012947","volume":"6","author":"Y Lin","year":"2021","unstructured":"Lin Y, McPhee J, Azad NL (2021) Comparison of deep reinforcement learning and model predictive control for adaptive cruise control. IEEE Trans Intell Veh 6(2):221\u2013231","journal-title":"IEEE Trans Intell Veh"},{"issue":"1","key":"11448_CR15","doi-asserted-by":"publisher","first-page":"453","DOI":"10.1109\/MITS.2022.3174410","volume":"15","author":"J Liu","year":"2023","unstructured":"Liu J, Li H, Yang Z, Dang S, Huang Z (2023) Deep dense network-based curriculum reinforcement learning for high-speed overtaking. IEEE Intell Transp Syst Mag 15(1):453\u2013466. https:\/\/doi.org\/10.1109\/MITS.2022.3174410","journal-title":"IEEE Intell Transp Syst Mag"},{"key":"11448_CR16","doi-asserted-by":"crossref","unstructured":"Lopez PA, Behrisch M, Bieker-Walz L, Erdmann J, Fl\u00f6tter\u00f6d YP, Hilbrich R, L\u00fccken L, Rummel J, Wagner P, Wie\u00dfner E (2018) Microscopic traffic simulation using sumo. In: International conference on intelligent transportation systems","DOI":"10.1109\/ITSC.2018.8569938"},{"key":"11448_CR17","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Rusu AA, Veness J, Bellemare MG, Graves A, Riedmiller MA, Fidjeland AK, Ostrovski G, Petersen S, Beattie C, Sadik A, Antonoglou I, King H, Kumaran D, Wierstra D, Legg S, Hassabis D (2015) Human-level control through deep reinforcement learning. Nature 518:529\u2013533","journal-title":"Nature"},{"key":"11448_CR18","unstructured":"Mnih V, Badia AP, Mirza M, Graves A, Lillicrap T, Harley T, Silver D, Kavukcuoglu K (2016) Asynchronous methods for deep reinforcement learning. In: Balcan MF, Weinberger KQ (eds.) Proceedings of the 33rd international conference on machine learning proceedings of machine learning research, vol. 48, pp 1928\u20131937. PMLR, New York, New York, USA . https:\/\/proceedings.mlr.press\/v48\/mniha16.html"},{"key":"11448_CR19","doi-asserted-by":"crossref","unstructured":"Moreno G, Nicolazzi LC, Vieira RDS, Martins D (2018) Stability of long combination vehicles. Int J Heavy Veh Syst 25","DOI":"10.1504\/IJHVS.2018.089897"},{"issue":"1","key":"11448_CR20","first-page":"7382","volume":"21","author":"S Narvekar","year":"2020","unstructured":"Narvekar S, Peng B, Leonetti M, Sinapov J, Taylor ME, Stone P (2020) Curriculum learning for reinforcement learning domains: a framework and survey. J Mach Learn Res 21(1):7382\u20137431","journal-title":"J Mach Learn Res"},{"key":"11448_CR21","doi-asserted-by":"publisher","first-page":"25","DOI":"10.1016\/j.trf.2018.02.004","volume":"55","author":"P Nilsson","year":"2018","unstructured":"Nilsson P, Laine L, Sandin J, Jacobson B, Eriksson O (2018) On actions of long combination vehicle drivers prior to lane changes in dense highway traffic - a driving simulator study. Transp Res F: Traffic Psychol Behav 55:25\u201337","journal-title":"Transp Res F: Traffic Psychol Behav"},{"key":"11448_CR22","doi-asserted-by":"publisher","unstructured":"Pathare D, Laine L, Chehreghani MH (2023) Improved tactical decision making and control architecture for autonomous truck in sumo using reinforcement learning. In: 2023 IEEE international conference on big data (BigData), pp 5321\u20135329 . https:\/\/doi.org\/10.1109\/BigData59044.2023.10386803","DOI":"10.1109\/BigData59044.2023.10386803"},{"issue":"268","key":"11448_CR23","first-page":"1","volume":"22","author":"A Raffin","year":"2021","unstructured":"Raffin A, Hill A, Gleave A, Kanervisto A, Ernestus M, Dormann N (2021) Stable-baselines3: reliable reinforcement learning implementations. J Mach Learn Res 22(268):1\u20138","journal-title":"J Mach Learn Res"},{"key":"11448_CR24","doi-asserted-by":"crossref","unstructured":"Sallab AE, Abdou M, Perot E, Yogamani S (2017) Deep reinforcement learning framework for autonomous driving. arXiv preprint arXiv:1704.02532","DOI":"10.2352\/ISSN.2470-1173.2017.19.AVM-023"},{"key":"11448_CR25","unstructured":"Schulman J, Wolski F, Dhariwal P, Radford A, Klimov O (2017) Proximal policy optimization algorithms"},{"key":"11448_CR26","unstructured":"Shalev-Shwartz S, Shammah S, Shashua A (2016) Safe, multi-agent, reinforcement learning for autonomous driving. arXiv preprint arXiv:1610.03295"},{"key":"11448_CR27","doi-asserted-by":"crossref","unstructured":"Shaout A, Colella D, Awad S (2011) Advanced driver assistance systems-past, present and future. In: International computer engineering conference. IEEE","DOI":"10.1109\/ICENCO.2011.6153935"},{"key":"11448_CR28","doi-asserted-by":"publisher","unstructured":"Song Y, Lin H, Kaufmann E, Durr P, Scaramuzza D (2021) Autonomous overtaking in gran turismo sport using curriculum reinforcement learning, pp 9403\u20139409. https:\/\/doi.org\/10.1109\/ICRA48506.2021.9561049","DOI":"10.1109\/ICRA48506.2021.9561049"},{"key":"11448_CR29","unstructured":"Sutton RS, Barto AG (2018) Reinforcement learning: an introduction, 2nd edn"},{"key":"11448_CR30","doi-asserted-by":"crossref","unstructured":"Svensson HG, Tyrchan C, Engkvist O, Chehreghani MH (2023) Utilizing reinforcement learning for de novo Drug Design 13(7):4811\u20134843","DOI":"10.1007\/s10994-024-06519-w"},{"issue":"2","key":"11448_CR31","doi-asserted-by":"publisher","first-page":"1805","DOI":"10.1103\/PhysRevE.62.1805","volume":"62","author":"M Treiber","year":"2000","unstructured":"Treiber M, Hennecke A, Helbing D (2000) Congested traffic states in empirical observations and microscopic simulations. Phys Rev E 62(2):1805","journal-title":"Phys Rev E"},{"issue":"10","key":"11448_CR32","doi-asserted-by":"publisher","first-page":"1167","DOI":"10.1080\/00423110903365910","volume":"48","author":"L Xiao","year":"2010","unstructured":"Xiao L, Gao F (2010) A comprehensive review of the development of adaptive cruise control systems. Veh Syst Dyn 48(10):1167\u20131192","journal-title":"Veh Syst Dyn"},{"key":"11448_CR34","doi-asserted-by":"crossref","unstructured":"Yang D, Qiu X, Yu D, Sun R, Pu Y (2015) A cellular automata model for car-truck heterogeneous traffic flow considering the car-truck following combination effect. Phys A: Stat Mech Appl 424","DOI":"10.1016\/j.physa.2014.12.020"},{"key":"11448_CR33","doi-asserted-by":"crossref","unstructured":"Yang J, Liu X, Liu S, Chu D, Lu L, Wu C (2020) Longitudinal tracking control of vehicle platooning using ddpg-based pid. In: 2020 4th CAA international conference on vehicular control and intelligence (CVCI)","DOI":"10.1109\/CVCI51460.2020.9338516"},{"issue":"11","key":"11448_CR35","doi-asserted-by":"publisher","first-page":"2089","DOI":"10.1007\/s00500-013-1110-y","volume":"17","author":"D Zhao","year":"2013","unstructured":"Zhao D, Wang B, Liu D (2013) A supervised actor-critic approach for adaptive cruise control. Soft Comput 17(11):2089\u20132099","journal-title":"Soft Comput"}],"container-title":["Artificial Intelligence Review"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-025-11448-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10462-025-11448-8","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-025-11448-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,27]],"date-time":"2026-01-27T03:09:16Z","timestamp":1769483356000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10462-025-11448-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,25]]},"references-count":35,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2026,1]]}},"alternative-id":["11448"],"URL":"https:\/\/doi.org\/10.1007\/s10462-025-11448-8","relation":{},"ISSN":["1573-7462"],"issn-type":[{"value":"1573-7462","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,11,25]]},"assertion":[{"value":"16 April 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 November 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 November 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"There is no confict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Confict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}},{"value":"Not applicable.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}}],"article-number":"27"}}