{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T16:24:00Z","timestamp":1777652640302,"version":"3.51.4"},"reference-count":33,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2022,8,12]],"date-time":"2022-08-12T00:00:00Z","timestamp":1660262400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,8,12]],"date-time":"2022-08-12T00:00:00Z","timestamp":1660262400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/100004839","name":"Department of Education","doi-asserted-by":"crossref","award":["ED#P116S210005"],"award-info":[{"award-number":["ED#P116S210005"]}],"id":[{"id":"10.13039\/100004839","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int. J. ITS Res."],"published-print":{"date-parts":[[2022,12]]},"DOI":"10.1007\/s13177-022-00321-5","type":"journal-article","created":{"date-parts":[[2022,8,12]],"date-time":"2022-08-12T07:03:03Z","timestamp":1660287783000},"page":"734-744","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":10,"title":["Intelligent Traffic Light via Policy-based Deep Reinforcement Learning"],"prefix":"10.1007","volume":"20","author":[{"given":"Yue","family":"Zhu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mingyu","family":"Cai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chris W.","family":"Schwarz","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junchao","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9658-7149","authenticated-orcid":false,"given":"Shaoping","family":"Xiao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,8,12]]},"reference":[{"key":"321_CR1","unstructured":"INRIX: Congestion Costs Each American 97 hours, $1,348 A Year - INRIX. https:\/\/inrix.com\/press-releases\/scorecard-2018-us\/. Accessed October 5, 2021."},{"key":"321_CR2","doi-asserted-by":"publisher","first-page":"307","DOI":"10.1016\/J.SCITOTENV.2013.01.074","volume":"450\u2013451","author":"K Zhang","year":"2013","unstructured":"Zhang, K., Batterman, S.: Air pollution and health risks due to vehicle traffic. Sci Total Environ. 450\u2013451, 307\u2013316 (2013). https:\/\/doi.org\/10.1016\/J.SCITOTENV.2013.01.074","journal-title":"Sci Total Environ."},{"key":"321_CR3","doi-asserted-by":"publisher","first-page":"3538","DOI":"10.1016\/J.TRPRO.2017.05.282","volume":"25","author":"S Bharadwaj","year":"2017","unstructured":"Bharadwaj, S., Ballare, S., Rohit, Chandel, M.K.: Impact of congestion on greenhouse gas emissions for road transport in Mumbai metropolitan region. Transp Res Procedia. 25, 3538\u20133551 (2017). https:\/\/doi.org\/10.1016\/J.TRPRO.2017.05.282","journal-title":"Transp Res Procedia."},{"issue":"4","key":"321_CR4","doi-asserted-by":"publisher","first-page":"386","DOI":"10.2307\/3006800","volume":"14","author":"AJ Miller","year":"1963","unstructured":"Miller, A.J.: Settings for Fixed-Cycle Traffic Signals. J Oper Res Soc. 14(4), 386 (1963). https:\/\/doi.org\/10.2307\/3006800","journal-title":"J Oper Res Soc."},{"key":"321_CR5","doi-asserted-by":"publisher","unstructured":"Cools, S.B., Gershenson, C., D\u2019Hooghe, B.: Self-Organizing Traffic Lights: A Realistic Simulation. In: Prokopenko M, ed. Advanced Information and Knowledge Processing. Springer, London; 45\u201355. (2013). https:\/\/doi.org\/10.1007\/978-1-4471-5113-5_3","DOI":"10.1007\/978-1-4471-5113-5_3"},{"key":"321_CR6","doi-asserted-by":"publisher","DOI":"10.1109\/VETECS.2011.5956434","author":"B Zhou","year":"2011","unstructured":"Zhou, B., Cao, J., Wu, H.: Adaptive traffic light control of multiple intersections in WSN-based ITS. IEEE Veh Technol Conf. (2011). https:\/\/doi.org\/10.1109\/VETECS.2011.5956434","journal-title":"IEEE Veh Technol Conf."},{"key":"321_CR7","doi-asserted-by":"publisher","first-page":"39897","DOI":"10.1109\/ACCESS.2021.3064310","volume":"9","author":"L Miao","year":"2021","unstructured":"Miao, L., Leitner, D.: Adaptive Traffic Light Control with Quality-of-Service Provisioning for Connected and Automated Vehicles at Isolated Intersections. IEEE Access. 9, 39897\u201339909 (2021). https:\/\/doi.org\/10.1109\/ACCESS.2021.3064310","journal-title":"IEEE Access."},{"key":"321_CR8","doi-asserted-by":"publisher","unstructured":"Dimitrov, S.: Optimal Control of Traffic Lights in Urban Area. 2020 Int Conf Autom Informatics, ICAI 2020 - Proc. October 2020. https:\/\/doi.org\/10.1109\/ICAI50593.2020.9311318","DOI":"10.1109\/ICAI50593.2020.9311318"},{"issue":"18","key":"321_CR9","doi-asserted-by":"publisher","first-page":"14359","DOI":"10.1007\/S00521-019-04480-7","volume":"32","author":"S Xiao","year":"2020","unstructured":"Xiao, S., Hu, R., Li, Z., Attarian, S., Bj\u00f6rk, K.-M., Lendasse, A.: A machine-learning-enhanced hierarchical multiscale method for bridging from molecular dynamics to continua. Neural Comput Appl. 32(18), 14359\u201314373 (2020). https:\/\/doi.org\/10.1007\/S00521-019-04480-7","journal-title":"Neural Comput Appl."},{"key":"321_CR10","doi-asserted-by":"crossref","unstructured":"Cai, M., Hasanbeig, M., Xiao, S., Abate, A., Kan, Z. Modular Deep Reinforcement Learning for Continuous Motion Planning with Temporal Logic. IEEE Robot. Autom. Lett. 6(4):7973\u20137980. (2021). http:\/\/arxiv.org\/abs\/2102.12855. Accessed April 8, 2021","DOI":"10.1109\/LRA.2021.3101544"},{"key":"321_CR11","volume-title":"Reinforcement Learning: An Introduction","author":"RS Sutton","year":"2018","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement Learning: An Introduction, 2nd edn. The MIT Press, London (2018)","edition":"2"},{"issue":"2","key":"321_CR12","doi-asserted-by":"publisher","first-page":"232","DOI":"10.1016\/S0377-2217(00)00123-5","volume":"131","author":"E Bingham","year":"2001","unstructured":"Bingham, E.: Reinforcement learning in neurofuzzy traffic signal control. Eur J Oper Res. 131(2), 232\u2013241 (2001). https:\/\/doi.org\/10.1016\/S0377-2217(00)00123-5","journal-title":"Eur J Oper Res."},{"key":"321_CR13","doi-asserted-by":"publisher","first-page":"656","DOI":"10.1007\/978-3-540-87479-9_61","volume":"5211","author":"L Kuyer","year":"2008","unstructured":"Kuyer, L., Whiteson, S., Bakker, B., Vlassis, N.: Multiagent Reinforcement Learning for Urban Traffic Control Using Coordination Graphs. Lect Notes Comput Sci. 5211, 656\u2013671 (2008). https:\/\/doi.org\/10.1007\/978-3-540-87479-9_61","journal-title":"Lect Notes Comput Sci."},{"issue":"3","key":"321_CR14","doi-asserted-by":"publisher","first-page":"247","DOI":"10.1109\/JAS.2016.7508798","volume":"3","author":"L Li","year":"2016","unstructured":"Li, L., Lv, Y., Wang, F.Y.: Traffic signal timing via deep reinforcement learning. IEEE\/CAA J Autom Sin. 3(3), 247\u2013254 (2016). https:\/\/doi.org\/10.1109\/JAS.2016.7508798","journal-title":"IEEE\/CAA J Autom Sin."},{"key":"321_CR15","doi-asserted-by":"publisher","unstructured":"Wei, H., Yao, H., Zheng, G., Li, Z.: IntelliLight: A reinforcement learning approach for intelligent traffic light control. Proc ACM SIGKDD Int Conf Knowl Discov Data Min. 2496\u20132505. (2018). https:\/\/doi.org\/10.1145\/3219819.3220096","DOI":"10.1145\/3219819.3220096"},{"issue":"8","key":"321_CR16","doi-asserted-by":"publisher","first-page":"8243","DOI":"10.1109\/TVT.2020.2997896","volume":"69","author":"T Wu","year":"2020","unstructured":"Wu, T., Zhou, P., Liu, K., et al.: Multi-Agent Deep Reinforcement Learning for Urban Traffic Light Control in Vehicular Networks. IEEE Trans Veh Technol. 69(8), 8243\u20138256 (2020). https:\/\/doi.org\/10.1109\/TVT.2020.2997896","journal-title":"IEEE Trans Veh Technol."},{"key":"321_CR17","doi-asserted-by":"publisher","unstructured":"Wang, Y., Xu, T., Niu, X., Tan, C., Chen, E., Xiong, H.: STMARL: A Spatio-Temporal Multi-Agent Reinforcement Learning Approach for Cooperative Traffic Light Control. IEEE Trans Mob Comput. 1\u20131 (2020). https:\/\/doi.org\/10.1109\/TMC.2020.3033782","DOI":"10.1109\/TMC.2020.3033782"},{"issue":"04","key":"321_CR18","doi-asserted-by":"publisher","first-page":"3414","DOI":"10.1609\/AAAI.V34I04.5744","volume":"34","author":"C Chen","year":"2020","unstructured":"Chen, C., Wei, H., Xu, N., et al.: Toward A Thousand Lights: Decentralized Deep Reinforcement Learning for Large-Scale Traffic Signal Control. Proc AAAI Conf Artif Intell. 34(04), 3414\u20133421 (2020). https:\/\/doi.org\/10.1609\/AAAI.V34I04.5744","journal-title":"Proc AAAI Conf Artif Intell."},{"key":"321_CR19","doi-asserted-by":"publisher","unstructured":"Wei, H., Xu, N., Zhang, H., et al.: Colight: Learning network-level cooperation for traffic signal control. Int Conf Inf Knowl Manag Proc. 1913\u20131922. (2019). https:\/\/doi.org\/10.1145\/3357384.3357902","DOI":"10.1145\/3357384.3357902"},{"key":"321_CR20","doi-asserted-by":"publisher","first-page":"2575","DOI":"10.1109\/ITSC.2018.8569938","volume":"2018","author":"PA Lopez","year":"2018","unstructured":"Lopez, P.A., Behrisch, M., Bieker-Walz, L., et al.: Microscopic Traffic Simulation using SUMO. IEEE Conf Intell Transp Syst Proceedings, ITSC. 2018, 2575\u20132582 (2018). https:\/\/doi.org\/10.1109\/ITSC.2018.8569938","journal-title":"IEEE Conf Intell Transp Syst Proceedings, ITSC."},{"issue":"2","key":"321_CR21","doi-asserted-by":"publisher","first-page":"1243","DOI":"10.1109\/TVT.2018.2890726","volume":"68","author":"X Liang","year":"2019","unstructured":"Liang, X., Du, X., Wang, G., Han, Z.: A Deep Reinforcement Learning Network for Traffic Light Cycle Control. IEEE Trans Veh Technol. 68(2), 1243\u20131253 (2019). https:\/\/doi.org\/10.1109\/TVT.2018.2890726","journal-title":"IEEE Trans Veh Technol."},{"key":"321_CR22","doi-asserted-by":"publisher","first-page":"877","DOI":"10.1109\/ITSC.2018.8569301","volume":"2018","author":"T Nishi","year":"2018","unstructured":"Nishi, T., Otaki, K., Hayakawa, K., Yoshimura, T.: Traffic Signal Control Based on Reinforcement Learning with Graph Convolutional Neural Nets. IEEE Conf Intell Transp Syst Proceedings, ITSC. 2018, 877\u2013883 (2018). https:\/\/doi.org\/10.1109\/ITSC.2018.8569301","journal-title":"IEEE Conf Intell Transp Syst Proceedings, ITSC."},{"key":"321_CR23","doi-asserted-by":"publisher","first-page":"279","DOI":"10.1007\/BF00992698","volume":"8","author":"C Watkins","year":"1992","unstructured":"Watkins, C., Dayan, P.: Q-Learning. Mach Learn. 8, 279\u2013292 (1992). https:\/\/doi.org\/10.1007\/BF00992698","journal-title":"Mach Learn."},{"key":"321_CR24","unstructured":"Mnih, V., Kavukcuoglu, K., Silver, D., et al.: Playing Atari with Deep Reinforcement Learning. https:\/\/arxiv.org\/abs\/1312.5602v1 (2013). Accessed September 19, 2021"},{"key":"321_CR25","doi-asserted-by":"crossref","unstructured":"Hasselt H van, Guez, A., Silver, D.: Deep Reinforcement Learning with Double Q-Learning. In: AAAI\u201916: Proceedings of the Thirtieth AAAI Conference on Artificial Intelligence. 30, 2094-2100 (2016)","DOI":"10.1609\/aaai.v30i1.10295"},{"issue":"11","key":"321_CR26","doi-asserted-by":"publisher","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y Lecun","year":"1998","unstructured":"Lecun, Y., Bottou, L., Bengio, Y., Haffner, P.: Gradient-based learning applied to document recognition. Proc IEEE. 86(11), 2278\u20132324 (1998). https:\/\/doi.org\/10.1109\/5.726791","journal-title":"Proc IEEE."},{"key":"321_CR27","unstructured":"Nair, V., Hinton, G.: Rectified linear units improve restricted boltzmann machines. In: Proceedings of the 27th International Conference on International Conference on Machine Learning. 32, 807\u2013814 (2010)"},{"issue":"3","key":"321_CR28","doi-asserted-by":"publisher","first-page":"293","DOI":"10.1007\/BF00992699","volume":"8","author":"LJ Lin","year":"1992","unstructured":"Lin, L.J.: Self-improving reactive agents based on reinforcement learning, planning and teaching. Mach Learn. 8(3), 293\u2013321 (1992). https:\/\/doi.org\/10.1007\/BF00992699","journal-title":"Mach Learn."},{"key":"321_CR29","unstructured":"Kakade, S., Langford, J.: Approximately optimal approximate reinforcement learning. In: In Proc. 19th International Conference on Machine Learning (2002)"},{"key":"321_CR30","unstructured":"Mnih, V., Badia, A.P., Mirza, M., et al.: Asynchronous Methods for Deep Reinforcement Learning. In: Balcan MF, Weinberger KQ, eds. Proceedings of The 33rd International Conference on Machine Learning. Vol 48. Proceedings of Machine Learning Research. New York, New York, USA: PMLR:1928\u20131937. https:\/\/proceedings.mlr.press\/v48\/mniha16.html (2016).\u00a0Accessed November 23, 2020"},{"key":"321_CR31","unstructured":"Schulman, J., Moritz, P., Levine, S., Jordan, M.I., Abbeel, P.: High-dimensional continuous control using generalized advantage estimation. In: 4th International Conference on Learning Representations, ICLR 2016 - Conference Track Proceedings. International Conference on Learning Representations, ICLR. https:\/\/sites.google.com\/site\/gaepapersupp (2016). Accessed November 23, 2020"},{"key":"321_CR32","unstructured":"Schulman, J., Levine, S., Moritz, P., Jordan, M.I., Abbeel, P.: Trust Region Policy Optimization. 32nd Int Conf Mach Learn ICML 2015. 3,1889\u20131897. http:\/\/arxiv.org\/abs\/1502.05477 (2015). Accessed November 23, 2020"},{"key":"321_CR33","unstructured":"Schulman J, Wolski F, Dhariwal P, Radford A, Klimov O. Proximal policy optimization algorithms. arXiv. https:\/\/arxiv.org\/abs\/1707.06347v2 (2017). Accessed November 23, 2020"}],"container-title":["International Journal of Intelligent Transportation Systems Research"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13177-022-00321-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13177-022-00321-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13177-022-00321-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,11,15]],"date-time":"2022-11-15T21:23:07Z","timestamp":1668547387000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13177-022-00321-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,8,12]]},"references-count":33,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2022,12]]}},"alternative-id":["321"],"URL":"https:\/\/doi.org\/10.1007\/s13177-022-00321-5","relation":{},"ISSN":["1348-8503","1868-8659"],"issn-type":[{"value":"1348-8503","type":"print"},{"value":"1868-8659","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,8,12]]},"assertion":[{"value":"15 January 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 June 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 July 2022","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 August 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not applicable.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics Approval and Consent to Participate"}},{"value":"All authors have approved the manuscript and gave their consent for submission and publication.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for Publication"}},{"value":"The authors declare no competing financial interests.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing Interests"}}]}}