{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,4]],"date-time":"2026-06-04T04:11:42Z","timestamp":1780546302532,"version":"3.54.1"},"reference-count":134,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2026,3,9]],"date-time":"2026-03-09T00:00:00Z","timestamp":1773014400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2026,3,27]],"date-time":"2026-03-27T00:00:00Z","timestamp":1774569600000},"content-version":"vor","delay-in-days":18,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["No.62406251"],"award-info":[{"award-number":["No.62406251"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Artif Intell Rev"],"DOI":"10.1007\/s10462-026-11530-9","type":"journal-article","created":{"date-parts":[[2026,3,9]],"date-time":"2026-03-09T15:47:35Z","timestamp":1773071255000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Intelligent traffic signal control based on reinforcement learning: a survey"],"prefix":"10.1007","volume":"59","author":[{"given":"Hang","family":"Xiao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9168-5038","authenticated-orcid":false,"given":"Huale","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhaobin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhen","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shuhan","family":"Qi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiajia","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"DingZhong","family":"Cai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"JiaQi","family":"Yin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,3,9]]},"reference":[{"key":"11530_CR1","doi-asserted-by":"publisher","first-page":"201","DOI":"10.1007\/s10489-013-0455-3","volume":"40","author":"M Abdoos","year":"2014","unstructured":"Abdoos M, Mozayani N, Bazzan AL (2014) Hierarchical control of traffic signals using q-learning with tile coding. Appl Intell 40:201\u2013213","journal-title":"Appl Intell"},{"issue":"3","key":"11530_CR2","doi-asserted-by":"publisher","first-page":"278","DOI":"10.1061\/(ASCE)0733-947X(2003)129:3(278)","volume":"129","author":"B Abdulhai","year":"2003","unstructured":"Abdulhai B, Pringle R, Karakoulas GJ (2003) Reinforcement learning for true adaptive traffic signal control. J Transp Eng 129(3):278\u2013285","journal-title":"J Transp Eng"},{"key":"11530_CR3","doi-asserted-by":"crossref","unstructured":"Araghi S, Khosravi A, Creighton D (2015) Distributed q-learning controller for a multi-intersection traffic network. In: Neural Information Processing: 22nd International Conference, ICONIP 2015, Istanbul, Turkey, November 9\u201312, 2015, Proceedings, Part I 22, pp. 337\u2013344. Springer","DOI":"10.1007\/978-3-319-26532-2_37"},{"issue":"2","key":"11530_CR4","doi-asserted-by":"publisher","first-page":"128","DOI":"10.1049\/iet-its.2009.0070","volume":"4","author":"I Arel","year":"2010","unstructured":"Arel I, Liu C, Urbanik T, Kohls AG (2010) Reinforcement learning-based multi-agent system for network traffic signal control. IET Intel Transport Syst 4(2):128\u2013135","journal-title":"IET Intel Transport Syst"},{"key":"11530_CR5","doi-asserted-by":"publisher","first-page":"732","DOI":"10.1016\/j.trc.2017.09.020","volume":"85","author":"M Aslani","year":"2017","unstructured":"Aslani M, Mesgari MS, Wiering M (2017) Adaptive traffic signal control with actor-critic methods in a real-world traffic network with different traffic disruption events. Transportation Research Part C Emerging Technologies 85:732\u2013752","journal-title":"Transportation Research Part C Emerging Technologies"},{"issue":"1","key":"11530_CR6","doi-asserted-by":"publisher","first-page":"40","DOI":"10.1080\/15472450.2017.1387546","volume":"22","author":"HA Aziz","year":"2018","unstructured":"Aziz HA, Zhu F, Ukkusuri SV (2018) Learning-based traffic signal control algorithms with neighborhood information sharing: An application for sustainable mobility. Journal of Intelligent Transportation Systems 22(1):40\u201352","journal-title":"Journal of Intelligent Transportation Systems"},{"key":"11530_CR7","doi-asserted-by":"publisher","unstructured":"Bakker B, Whiteson S, Kester L, Groen FCA (2010) Traffic Light Control by Multiagent Reinforcement Learning Systems, pp. 475\u2013510. Springer, Berlin, Heidelberg . https:\/\/doi.org\/10.1007\/978-3-642-11688-9_18","DOI":"10.1007\/978-3-642-11688-9_18"},{"issue":"3","key":"11530_CR8","doi-asserted-by":"publisher","first-page":"177","DOI":"10.1049\/iet-its.2009.0096","volume":"4","author":"P Balaji","year":"2010","unstructured":"Balaji P, German X, Srinivasan D (2010) Urban traffic signal control using reinforcement learning agents. IET Intel Transport Syst 4(3):177\u2013188","journal-title":"IET Intel Transport Syst"},{"key":"11530_CR9","doi-asserted-by":"crossref","unstructured":"Bellman R (1957) A markovian decision process. Journal of mathematics and mechanics, 679\u2013684","DOI":"10.1512\/iumj.1957.6.56038"},{"issue":"1","key":"11530_CR10","doi-asserted-by":"publisher","first-page":"65","DOI":"10.1080\/09540091.2014.885282","volume":"26","author":"T Brys","year":"2014","unstructured":"Brys T, Pham TT, Taylor ME (2014) Distributed learning and multi-objectivity in traffic light control. Connect Sci 26(1):65\u201383","journal-title":"Connect Sci"},{"key":"11530_CR11","unstructured":"Calvo JA, Dusparic I (2018) Heterogeneous multi-agent deep reinforcement learning for traffic lights control. In: AICS, pp. 2\u201313"},{"key":"11530_CR12","doi-asserted-by":"crossref","unstructured":"Camponogara E, Kraus Jr W (2003) Distributed learning agents in urban traffic control. In: Portuguese Conference on Artificial Intelligence, pp. 324\u2013335. Springer","DOI":"10.1007\/978-3-540-24580-3_38"},{"issue":"10","key":"11530_CR13","doi-asserted-by":"publisher","first-page":"17899","DOI":"10.1109\/TITS.2022.3159714","volume":"23","author":"M Cao","year":"2022","unstructured":"Cao M, Li VO, Shuai Q (2022) A gain with no pain: Exploring intelligent traffic signal control for emergency vehicles. IEEE Trans Intell Transp Syst 23(10):17899\u201317909","journal-title":"IEEE Trans Intell Transp Syst"},{"issue":"7","key":"11530_CR14","doi-asserted-by":"publisher","first-page":"6836","DOI":"10.1109\/TITS.2023.3257199","volume":"24","author":"M Cao","year":"2023","unstructured":"Cao M, Li VO, Shuai Q (2023) Deepgal: Intelligent vehicle control for traffic congestion alleviation at intersections. IEEE Trans Intell Transp Syst 24(7):6836\u20136848","journal-title":"IEEE Trans Intell Transp Syst"},{"key":"11530_CR15","unstructured":"Casas N (2017) Deep deterministic policy gradient for urban traffic light control. arXiv preprint arXiv:1703.09035"},{"key":"11530_CR16","doi-asserted-by":"crossref","unstructured":"Chen H, Jiang Y, Guo S, Mao X, Lin Y, Wan H (2024) Difflight: A partial rewards conditioned diffusion model for traffic signal control with missing data. arXiv preprint arXiv:2410.22938","DOI":"10.52202\/079017-3921"},{"key":"11530_CR17","doi-asserted-by":"crossref","unstructured":"Chen C, Wei H, Xu N, Zheng G, Yang M, Xiong Y, Xu K, Li Z (2020) Toward a thousand lights: Decentralized deep reinforcement learning for large-scale traffic signal control. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, pp. 3414\u20133421","DOI":"10.1609\/aaai.v34i04.5744"},{"key":"11530_CR18","doi-asserted-by":"crossref","unstructured":"Chen W, Yang S, Li W, Hu Y, Liu X, Gao Y (2024) Learning multi-intersection traffic signal control via coevolutionary multi-agent reinforcement learning. IEEE Transactions on Intelligent Transportation Systems","DOI":"10.36227\/techrxiv.23254547"},{"key":"11530_CR19","doi-asserted-by":"crossref","unstructured":"Chin YK, Lee LK, Bolong N, Yang SS, Teo KTK (2011) Exploring q-learning optimization in traffic signal timing plan management. In: 2011 Third International Conference on Computational Intelligence, Communication Systems and Networks, pp. 269\u2013274. IEEE","DOI":"10.1109\/CICSyN.2011.64"},{"key":"11530_CR20","doi-asserted-by":"publisher","DOI":"10.2172\/885576","volume-title":"Temporary losses of highway capacity and impacts on performance: Phase 2","author":"S-M Chin","year":"2004","unstructured":"Chin S-M, Franzese O, Greene DL, Hwang H-L, Gibson R et al (2004) Temporary losses of highway capacity and impacts on performance: Phase 2. United States. Dept. of Energy. Office of Scientific and Technical Information, Technical report"},{"key":"11530_CR21","doi-asserted-by":"crossref","unstructured":"Choe C-J, Baek S, Woon B, Kong S-H (2018) Deep q learning with lstm for traffic light control. In: 2018 24th Asia-Pacific Conference on Communications (APCC), pp. 331\u2013336. IEEE","DOI":"10.1109\/APCC.2018.8633520"},{"issue":"3","key":"11530_CR22","doi-asserted-by":"publisher","first-page":"1086","DOI":"10.1109\/TITS.2019.2901791","volume":"21","author":"T Chu","year":"2019","unstructured":"Chu T, Wang J, Codec\u00e0 L, Li Z (2019) Multi-agent deep reinforcement learning for large-scale traffic signal control. IEEE Trans Intell Transp Syst 21(3):1086\u20131095","journal-title":"IEEE Trans Intell Transp Syst"},{"key":"11530_CR23","doi-asserted-by":"crossref","unstructured":"Chu T, Qu S, Wang J (2016) Large-scale traffic grid signal control with regional reinforcement learning. In: 2016 American Control Conference (acc), pp. 815\u2013820. IEEE","DOI":"10.1109\/ACC.2016.7525014"},{"key":"11530_CR24","doi-asserted-by":"crossref","unstructured":"Cools S-B, Gershenson C, D\u2019Hooghe B (2013) Self-organizing traffic lights: A realistic simulation. Advances in applied self-organizing systems, 45\u201355","DOI":"10.1007\/978-1-4471-5113-5_3"},{"key":"11530_CR25","doi-asserted-by":"crossref","unstructured":"Da L, Gao M, Mei H, Wei H (2024) Prompt to transfer: Sim-to-real transfer for traffic signal control with prompt learning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, pp. 82\u201390","DOI":"10.1609\/aaai.v38i1.27758"},{"issue":"17\u201318","key":"11530_CR26","doi-asserted-by":"publisher","first-page":"8563","DOI":"10.1007\/s10489-024-05637-1","volume":"54","author":"X Deng","year":"2024","unstructured":"Deng X, Yin S, Pei X, Lin L, Chen X, Gui J (2024) E-dbrl: efficient double broad reinforcement learning for adaptive traffic signal control. Appl Intell 54(17\u201318):8563\u20138575. https:\/\/doi.org\/10.1007\/s10489-024-05637-1","journal-title":"Appl Intell"},{"issue":"7","key":"11530_CR27","doi-asserted-by":"publisher","first-page":"7496","DOI":"10.1109\/TITS.2021.3070835","volume":"23","author":"F-X Devailly","year":"2021","unstructured":"Devailly F-X, Larocque D, Charlin L (2021) Ig-rl: Inductive graph reinforcement learning for massive-scale traffic signal control. IEEE Trans Intell Transp Syst 23(7):7496\u20137507","journal-title":"IEEE Trans Intell Transp Syst"},{"key":"11530_CR28","doi-asserted-by":"crossref","unstructured":"Du W, Ye J, Gu J, Li J, Wei H, Wang G (2023) Safelight: A reinforcement learning method toward collision-free traffic signal control. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 37, pp. 14801\u201314810","DOI":"10.1609\/aaai.v37i12.26729"},{"issue":"3","key":"11530_CR29","doi-asserted-by":"publisher","first-page":"1140","DOI":"10.1109\/TITS.2013.2255286","volume":"14","author":"S El-Tantawy","year":"2013","unstructured":"El-Tantawy S, Abdulhai B, Abdelgawad H (2013) Multiagent reinforcement learning for integrated network of adaptive traffic signal controllers (marlin-atsc): methodology and large-scale application on downtown toronto. IEEE Trans Intell Transp Syst 14(3):1140\u20131150","journal-title":"IEEE Trans Intell Transp Syst"},{"issue":"3","key":"11530_CR30","doi-asserted-by":"publisher","first-page":"227","DOI":"10.1080\/15472450.2013.810991","volume":"18","author":"S El-Tantawy","year":"2014","unstructured":"El-Tantawy S, Abdulhai B, Abdelgawad H (2014) Design of reinforcement learning parameters for seamless application of adaptive traffic signal control. Journal of Intelligent Transportation Systems 18(3):227\u2013245","journal-title":"Journal of Intelligent Transportation Systems"},{"key":"11530_CR31","doi-asserted-by":"crossref","unstructured":"El-Tantawy S, Abdulhai B (2010) An agent-based learning towards decentralized and coordinated traffic signal control. In: 13th International IEEE Conference on Intelligent Transportation Systems, pp. 665\u2013670. IEEE","DOI":"10.1109\/ITSC.2010.5625066"},{"key":"11530_CR32","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1186\/s12544-020-00440-8","volume":"12","author":"M Eom","year":"2020","unstructured":"Eom M, Kim B-I (2020) The traffic signal control problem for intersections: a review. Eur Transp Res Rev 12:1\u201320","journal-title":"Eur Transp Res Rev"},{"key":"11530_CR33","unstructured":"Gao J, Shen Y, Liu J, Ito M, Shiratori N (2017) Adaptive traffic signal control: Deep reinforcement learning algorithm with experience replay and target network. arXiv preprint arXiv:1705.02755"},{"key":"11530_CR34","doi-asserted-by":"crossref","unstructured":"Garg D, Chli M, Vogiatzis G (2018) Deep reinforcement learning for autonomous traffic light control. In: 2018 3rd IEEE International Conference on Intelligent Transportation Engineering (ICITE), pp. 214\u2013218. IEEE","DOI":"10.1109\/ICITE.2018.8492537"},{"key":"11530_CR35","doi-asserted-by":"publisher","first-page":"26","DOI":"10.1016\/j.procs.2018.04.008","volume":"130","author":"W Genders","year":"2018","unstructured":"Genders W, Razavi S (2018) Evaluating reinforcement learning state representations for adaptive traffic signal control. Procedia computer science 130:26\u201333","journal-title":"Procedia computer science"},{"issue":"4","key":"11530_CR36","doi-asserted-by":"publisher","first-page":"319","DOI":"10.1080\/15472450.2018.1491003","volume":"23","author":"W Genders","year":"2019","unstructured":"Genders W, Razavi S (2019) Asynchronous n-step q-learning adaptive traffic signal control. Journal of Intelligent Transportation Systems 23(4):319\u2013331","journal-title":"Journal of Intelligent Transportation Systems"},{"key":"11530_CR37","unstructured":"Genders W, Razavi S (2016) Using a deep reinforcement learning agent for traffic signal control. arXiv preprint arXiv:1611.01142"},{"key":"11530_CR38","doi-asserted-by":"crossref","unstructured":"Goel H, Zhang Y, Damani M, Sartoretti G (2023) Sociallight: Distributed cooperation learning towards network-wide traffic signal control. arXiv preprint arXiv:2305.16145","DOI":"10.65109\/GIFG9402"},{"issue":"10","key":"11530_CR39","doi-asserted-by":"publisher","first-page":"10501","DOI":"10.1109\/TITS.2023.3276416","volume":"24","author":"J Guo","year":"2023","unstructured":"Guo J, Cheng L, Wang S (2023) Cotv: Cooperative control for traffic light signals and connected autonomous vehicles using deep reinforcement learning. IEEE Trans Intell Transp Syst 24(10):10501\u201310512","journal-title":"IEEE Trans Intell Transp Syst"},{"key":"11530_CR40","doi-asserted-by":"crossref","unstructured":"Gu H, Wang S, Ma X, Jia D, Mao G, Lim EG, Wong CPR (2024) Large-scale traffic signal control using constrained network partition and adaptive deep reinforcement learning. IEEE Transactions on Intelligent Transportation Systems","DOI":"10.1109\/TITS.2024.3352446"},{"key":"11530_CR41","doi-asserted-by":"crossref","unstructured":"Han T, Oguchi T, Lyu S (2022) Queuelearner: A knowledge-combined reinforcement learning to understand queuing evolution in isolated traffic signal control. In: 2022 IEEE 25th International Conference on Intelligent Transportation Systems (ITSC), pp. 1175\u20131182. IEEE","DOI":"10.1109\/ITSC55140.2022.9922584"},{"key":"11530_CR42","doi-asserted-by":"crossref","unstructured":"Han X, Zhao X, Zhang L, Wang W (2023) Mitigating action hysteresis in traffic signal control with traffic predictive reinforcement learning. In: Proceedings of the 29th ACM SIGKDD Conference on Knowledge Discovery and Data Mining, pp. 673\u2013684","DOI":"10.1145\/3580305.3599528"},{"key":"11530_CR43","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1155\/2010\/724035","volume":"2010","author":"D Houli","year":"2010","unstructured":"Houli D, Zhiheng L, Yi Z (2010) Multiobjective reinforcement learning for traffic signal control using vehicular ad hoc network. EURASIP journal on advances in signal processing 2010:1\u20137","journal-title":"EURASIP journal on advances in signal processing"},{"key":"11530_CR44","doi-asserted-by":"crossref","unstructured":"Huang H, Hu Z, Wang Y, Lu Z, Wen X (2023) Intersec2vec-tsc: Intersection representation learning for large-scale traffic signal control. IEEE Transactions on Intelligent Transportation Systems","DOI":"10.1109\/TITS.2023.3340153"},{"key":"11530_CR45","unstructured":"I\u0161a J, Kooij J, Koppejan R, Kuijer L (2006) Reinforcement learning of traffic light controllers adapting to accidents. Design and Organisation of Autonomous Systems, 1\u201314"},{"key":"11530_CR46","unstructured":"Jiang H, Li Z, Wei H, Xiong X, Ruan J, Lu J, Mao H, Zhao R (2024) X-light: Cross-city traffic signal control using transformer on transformer as meta multi-agent reinforcement learner. arXiv preprint arXiv:2404.12090"},{"key":"11530_CR47","doi-asserted-by":"crossref","unstructured":"Jiang Q, Qin M, Shi S, Sun W, Zheng B (2022) Multi-agent reinforcement learning for traffic signal control through universal communication method. arXiv preprint arXiv:2204.12190","DOI":"10.24963\/ijcai.2022\/535"},{"key":"11530_CR48","doi-asserted-by":"crossref","unstructured":"Jiang Q, Qin M, Zhang H, Zhang X, Sun W (2024) Blindlight: High robustness reinforcement learning method to solve partially blinded traffic signal control problem. IEEE Transactions on Intelligent Transportation Systems","DOI":"10.1109\/TITS.2024.3416154"},{"issue":"10","key":"11530_CR49","doi-asserted-by":"publisher","first-page":"3900","DOI":"10.1109\/TITS.2019.2906260","volume":"20","author":"J Jin","year":"2019","unstructured":"Jin J, Ma X (2019) A multi-objective agent-based control approach with application in intelligent traffic signal system. IEEE Trans Intell Transp Syst 20(10):3900\u20133912","journal-title":"IEEE Trans Intell Transp Syst"},{"key":"11530_CR50","doi-asserted-by":"crossref","unstructured":"Khamis MA, Gomaa W (2012) Enhanced multiagent multi-objective reinforcement learning for urban traffic light control. In: 2012 11th International Conference on Machine Learning and Applications, vol. 1, pp. 586\u2013591. IEEE","DOI":"10.1109\/ICMLA.2012.108"},{"key":"11530_CR51","doi-asserted-by":"crossref","unstructured":"Khamis MA, Gomaa W, El-Shishiny H (2012) Multi-objective traffic light control system based on bayesian probability interpretation. In: 2012 15th International IEEE Conference on Intelligent Transportation Systems, pp. 995\u20131000. IEEE","DOI":"10.1109\/ITSC.2012.6338853"},{"key":"11530_CR52","doi-asserted-by":"publisher","first-page":"134","DOI":"10.1016\/j.engappai.2014.01.007","volume":"29","author":"MA Khamis","year":"2014","unstructured":"Khamis MA, Gomaa W (2014) Adaptive multi-objective reinforcement learning with hybrid exploration for traffic signal control based on cooperative multi-agent framework. Eng Appl Artif Intell 29:134\u2013151","journal-title":"Eng Appl Artif Intell"},{"key":"11530_CR53","doi-asserted-by":"crossref","unstructured":"Kong AY, Lu BX, Yang CZ, Zhang DM (2022) A deep reinforcement learning framework with memory network to coordinate traffic signal control. In: 2022 IEEE 25th International Conference on Intelligent Transportation Systems (ITSC), pp. 3825\u20133830. IEEE","DOI":"10.1109\/ITSC55140.2022.9921752"},{"key":"11530_CR54","volume-title":"Traffic signal timing manual","author":"P Koonce","year":"2008","unstructured":"Koonce P et al (2008) Traffic signal timing manual. United States. Federal Highway Administration, Technical report"},{"key":"11530_CR55","doi-asserted-by":"crossref","unstructured":"Kunjir M, Chawla S, Chandrasekar S, Jay D, Ravindran B (2023) Optimizing traffic control with model-based learning: A pessimistic approach to data-efficient policy inference. In: Proceedings of the 29th ACM SIGKDD Conference on Knowledge Discovery and Data Mining, pp. 1176\u20131187","DOI":"10.1145\/3580305.3599459"},{"key":"11530_CR56","doi-asserted-by":"crossref","unstructured":"Kuyer L, Whiteson S, Bakker B, Vlassis N (2008) Multiagent reinforcement learning for urban traffic control using coordination graphs. In: Machine Learning and Knowledge Discovery in Databases: European Conference, ECML PKDD 2008, Antwerp, Belgium, September 15\u201319, 2008, Proceedings, Part I 19, pp. 656\u2013671. Springer","DOI":"10.1007\/978-3-540-87479-9_61"},{"issue":"3","key":"11530_CR57","doi-asserted-by":"publisher","first-page":"247","DOI":"10.1109\/JAS.2016.7508798","volume":"3","author":"L Li","year":"2016","unstructured":"Li L, Lv Y, Wang F-Y (2016) Traffic signal timing via deep reinforcement learning. IEEE\/CAA Journal of Automatica Sinica 3(3):247\u2013254","journal-title":"IEEE\/CAA Journal of Automatica Sinica"},{"issue":"2","key":"11530_CR58","doi-asserted-by":"publisher","first-page":"1243","DOI":"10.1109\/TVT.2018.2890726","volume":"68","author":"X Liang","year":"2019","unstructured":"Liang X, Du X, Wang G, Han Z (2019) A deep reinforcement learning network for traffic light cycle control. IEEE Trans Veh Technol 68(2):1243\u20131253","journal-title":"IEEE Trans Veh Technol"},{"key":"11530_CR59","doi-asserted-by":"crossref","unstructured":"Liang E, Su Z, Fang C, Zhong R (2022) Oam: An option-action reinforcement learning framework for universal multi-intersection control. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 36, pp. 4550\u20134558","DOI":"10.1609\/aaai.v36i4.20378"},{"key":"11530_CR60","doi-asserted-by":"crossref","unstructured":"Li M, Hu Z, Huang H, Lu Z, Wen X (2022) A hierarchical spatio-temporal cooperative reinforcement learning approach for traffic signal control. In: 2022 IEEE 25th International Conference on Intelligent Transportation Systems (ITSC), pp. 3411\u20133416. IEEE","DOI":"10.1109\/ITSC55140.2022.9922065"},{"key":"11530_CR61","unstructured":"Li J, Lin S, Shi T, Tian C, Mei Y, Song J, Zhan X, Li R (2023) A Fully Data-Driven Approach for Realistic Traffic Signal Control Using Offline Reinforcement Learning. https:\/\/arxiv.org\/abs\/2311.15920"},{"key":"11530_CR62","doi-asserted-by":"crossref","unstructured":"Li L, Li R, Peng Y, Huang C, Yuan J (2022) Cooperative max-pressure enhanced traffic signal control. In: Proceedings of the 31st ACM International Conference on Information & Knowledge Management, pp. 4173\u20134177","DOI":"10.1145\/3511808.3557569"},{"key":"11530_CR63","unstructured":"Lillicrap TP, Hunt JJ, Pritzel A, Heess NMO, Erez T, Tassa Y, Silver D, Wierstra D (2015) Continuous control with deep reinforcement learning. CoRR abs\/1509.02971"},{"key":"11530_CR64","unstructured":"Lin Y, Dai X, Li L, Wang F-Y (2018) An efficient deep reinforcement learning model for urban traffic control. arXiv preprint arXiv:1808.01876"},{"key":"11530_CR65","unstructured":"Little JDC, Kelson MD, Gartner NH (1980) MAXBAND: A Versatile Program for Setting Signals on Arteries and Triangular Networks. Working paper (Alfred P. Sloan School of Management). Alfred P. Sloan School of Management, Massachusetts Institute of Technology, ???. https:\/\/books.google.co.jp\/books?id=jqAZtwAACAAJ"},{"key":"11530_CR66","doi-asserted-by":"crossref","unstructured":"Liu Y, Luo G, Yuan Q, Li J, Jin L, Chen B, Pan R (2023) Gplight: Grouped multi-agent reinforcement learning for large-scale traffic signal control. In: IJCAI, pp. 199\u2013207","DOI":"10.24963\/ijcai.2023\/23"},{"issue":"8","key":"11530_CR67","doi-asserted-by":"publisher","first-page":"5675","DOI":"10.1007\/s10462-020-09831-8","volume":"53","author":"A Louati","year":"2020","unstructured":"Louati A (2020) A hybridization of deep learning techniques to predict and control traffic disturbances. Artif Intell Rev 53(8):5675\u20135704","journal-title":"Artif Intell Rev"},{"key":"11530_CR68","doi-asserted-by":"publisher","DOI":"10.1016\/j.ins.2024.121485","volume":"689","author":"Y Lu","year":"2025","unstructured":"Lu Y, Hegyi A, Maria Salomons A, Wang H (2025) Reference rl: Reinforcement learning with reference mechanism and its application in traffic signal control. Inf Sci 689:121485. https:\/\/doi.org\/10.1016\/j.ins.2024.121485","journal-title":"Inf Sci"},{"key":"11530_CR69","doi-asserted-by":"crossref","unstructured":"Lu J, Ruan J, Jiang H, Li Z, Mao H, Zhao R (2023) Dualight: Enhancing traffic signal control by leveraging scenario-specific and scenario-shared knowledge. arXiv preprint arXiv:2312.14532","DOI":"10.65109\/KEWS9599"},{"key":"11530_CR70","doi-asserted-by":"crossref","unstructured":"Ma YJ, Liang W, Wang H, Wang S, Zhu Y, Fan L, Bastani O, Jayaraman D (2024) Dreureka: Language model guided sim-to-real transfer. In: Robotics: Science and Systems (RSS)","DOI":"10.15607\/RSS.2024.XX.094"},{"issue":"3","key":"11530_CR71","doi-asserted-by":"publisher","first-page":"3129","DOI":"10.1109\/TITS.2022.3229477","volume":"24","author":"F Mao","year":"2022","unstructured":"Mao F, Li Z, Lin Y, Li L (2022) Mastering arterial traffic signal control with multi-agent attention-based soft actor-critic model. IEEE Trans Intell Transp Syst 24(3):3129\u20133144","journal-title":"IEEE Trans Intell Transp Syst"},{"key":"11530_CR72","doi-asserted-by":"crossref","unstructured":"Ma J, Wu F (2020) Feudal multi-agent deep reinforcement learning for traffic signal control. In: Proceedings of the 19th International Conference on Autonomous Agents and Multiagent Systems (AAMAS), pp. 816\u2013824","DOI":"10.65109\/UXZA9095"},{"key":"11530_CR73","unstructured":"McShane WR, Roess RP, Prassas ES (1998) Traffic Engineering. Bibliyografya Ve Indeks. Prentice Hall, ??? . https:\/\/books.google.co.jp\/books?id=EGsnAQAAMAAJ"},{"key":"11530_CR74","doi-asserted-by":"crossref","unstructured":"Mikami S, Kakazu Y (1994) Genetic reinforcement learning for cooperative traffic signal control. In: Proceedings of the First IEEE Conference on Evolutionary Computation. IEEE World Congress on Computational Intelligence, pp. 223\u2013228. IEEE","DOI":"10.1109\/ICEC.1994.350012"},{"issue":"10","key":"11530_CR75","doi-asserted-by":"publisher","first-page":"1269","DOI":"10.1049\/itr2.12208","volume":"16","author":"M Mileti\u0107","year":"2022","unstructured":"Mileti\u0107 M, Ivanjko E, Greguri\u0107 M, Ku\u0161i\u0107 K (2022) A review of reinforcement learning applications in adaptive traffic signal control. IET Intel Transport Syst 16(10):1269\u20131285","journal-title":"IET Intel Transport Syst"},{"issue":"7540","key":"11530_CR76","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Rusu AA, Veness J, Bellemare MG, Graves A, Riedmiller M, Fidjeland AK, Ostrovski G et al (2015) Human-level control through deep reinforcement learning. nature 518(7540):529\u2013533","journal-title":"nature"},{"key":"11530_CR77","unstructured":"Mnih V, Badia AP, Mirza M, Graves A, Lillicrap TP, Harley T, Silver D, Kavukcuoglu K (2016) Asynchronous methods for deep reinforcement learning. In: International Conference on Machine Learning. https:\/\/api.semanticscholar.org\/CorpusID:6875312"},{"key":"11530_CR78","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Graves A, Antonoglou I, Wierstra D, Riedmiller M (2013) Playing atari with deep reinforcement learning. Computer Science"},{"issue":"7","key":"11530_CR79","doi-asserted-by":"publisher","first-page":"417","DOI":"10.1049\/iet-its.2017.0153","volume":"11","author":"SS Mousavi","year":"2017","unstructured":"Mousavi SS, Schukat M, Howley E (2017) Traffic light control using deep policy-gradient and value-function-based reinforcement learning. IET Intel Transport Syst 11(7):417\u2013423","journal-title":"IET Intel Transport Syst"},{"key":"11530_CR80","doi-asserted-by":"crossref","unstructured":"M\u00fcller A, Sabatelli M (2022) Safe and psychologically pleasant traffic signal control with reinforcement learning using action masking. In: 2022 IEEE 25th International Conference on Intelligent Transportation Systems (ITSC), pp. 951\u2013958 . IEEE","DOI":"10.1109\/ITSC55140.2022.9922306"},{"key":"11530_CR81","doi-asserted-by":"publisher","unstructured":"M\u00fcller A, Sabatelli M (2023) Bridging the reality gap of reinforcement learning based traffic signal control using domain randomization and meta learning. In: 2023 IEEE 26th International Conference on Intelligent Transportation Systems (ITSC), pp. 5271\u20135278. https:\/\/doi.org\/10.1109\/ITSC57777.2023.10421987","DOI":"10.1109\/ITSC57777.2023.10421987"},{"key":"11530_CR82","doi-asserted-by":"crossref","unstructured":"Nishi T, Otaki K, Hayakawa K, Yoshimura T (2018) Traffic signal control based on reinforcement learning with graph convolutional neural nets. In: 2018 21st International Conference on Intelligent Transportation Systems (ITSC), pp. 877\u2013883. IEEE","DOI":"10.1109\/ITSC.2018.8569301"},{"key":"11530_CR83","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2022.116830","volume":"199","author":"M Noaeen","year":"2022","unstructured":"Noaeen M, Naik A, Goodman L, Crebo J, Abrar T, Abad ZSH, Bazzan AL, Far B (2022) Reinforcement learning in urban network traffic signal control: A systematic literature review. Expert Syst Appl 199:116830","journal-title":"Expert Syst Appl"},{"key":"11530_CR84","unstructured":"Pearl J (2012) The do-calculus revisited. In: Conference on Uncertainty in Artificial Intelligence. https:\/\/api.semanticscholar.org\/CorpusID:2768684"},{"key":"11530_CR85","unstructured":"Pham TT, Brys T, Taylor ME, Brys T, Drugan MM, Bosman P, Cock M-D, Lazar C, Demarchi L, Steenhoff D, et al (2013) Learning coordinated traffic light control. In: Proceedings of the Adaptive and Learning Agents Workshop (at AAMAS-13), vol. 10, pp. 1196\u20131201. IEEE"},{"key":"11530_CR86","first-page":"21","volume":"8","author":"E Pol","year":"2016","unstructured":"Pol E, Oliehoek FA (2016) Coordinated deep reinforcement learners for traffic light control. Proceedings of learning inference and control of multi-agent systems (at NIPS 2016) 8:21\u201338","journal-title":"Proceedings of learning inference and control of multi-agent systems (at NIPS 2016)"},{"issue":"2","key":"11530_CR87","first-page":"412","volume":"12","author":"L Prashanth","year":"2010","unstructured":"Prashanth L, Bhatnagar S (2010) Reinforcement learning with function approximation for traffic signal control. IEEE Trans Intell Transp Syst 12(2):412\u2013421","journal-title":"IEEE Trans Intell Transp Syst"},{"key":"11530_CR88","doi-asserted-by":"crossref","unstructured":"Prashanth L, Bhatnagar S (2011) Reinforcement learning with average cost for adaptive control of traffic lights at intersections. In: 2011 14th International IEEE Conference on Intelligent Transportation Systems (ITSC), pp. 1640\u20131645. IEEE","DOI":"10.1109\/ITSC.2011.6082823"},{"key":"11530_CR89","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1186\/s12544-020-00439-1","volume":"12","author":"SSSM Qadri","year":"2020","unstructured":"Qadri SSSM, G\u00f6k\u00e7e MA, \u00d6ner E (2020) State-of-art review of traffic signal control methods: challenges and opportunities. Eur Transp Res Rev 12:1\u201323","journal-title":"Eur Transp Res Rev"},{"key":"11530_CR90","doi-asserted-by":"crossref","unstructured":"Ren Y, Wu J, Yi C, Ran Y, Lou Y (2022) Meta-reinforcement learning for centralized multiple intersections traffic signal control. In: 2022 IEEE 25th International Conference on Intelligent Transportation Systems (ITSC), pp. 281\u2013286. IEEE","DOI":"10.1109\/ITSC55140.2022.9922355"},{"key":"11530_CR91","doi-asserted-by":"crossref","unstructured":"Ruan J, Li Z, Wei H, Jiang H, Lu J, Xiong X, Mao H, Zhao R (2024) Coslight: Co-optimizing collaborator selection and decision-making to enhance traffic signal control. In: Proceedings of the 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining, pp. 2500\u20132511","DOI":"10.1145\/3637528.3671998"},{"key":"11530_CR92","doi-asserted-by":"crossref","unstructured":"Shabestary SMA, Abdulhai B (2018) Deep learning vs. discrete reinforcement learning for adaptive traffic signal control. In: 2018 21st International Conference on Intelligent Transportation Systems (ITSC), pp. 286\u2013293. IEEE","DOI":"10.1109\/ITSC.2018.8569549"},{"issue":"3","key":"11530_CR93","doi-asserted-by":"publisher","first-page":"1","DOI":"10.9734\/JAMCS\/2018\/41281","volume":"27","author":"S Shi","year":"2018","unstructured":"Shi S, Chen F (2018) Deep recurrent q-learning method for area traffic coordination control. Journal of Advances in Mathematics and Computer Science 27(3):1\u201311","journal-title":"Journal of Advances in Mathematics and Computer Science"},{"key":"11530_CR94","doi-asserted-by":"crossref","unstructured":"Shoufeng L, Ximin L, Shiqiang D (2008) Q-learning for adaptive traffic signal control based on delay minimization strategy. In: 2008 IEEE International Conference on Networking, Sensing and Control, pp. 687\u2013691. IEEE","DOI":"10.1109\/ICNSC.2008.4525304"},{"key":"11530_CR95","unstructured":"Steingrover M, Schouten R, Peelen S, Nijhuis E, Bakker B et al (2005) Reinforcement learning of traffic light controllers adapting to traffic congestion. In: BNAIC, pp. 216\u2013223"},{"key":"11530_CR96","doi-asserted-by":"publisher","first-page":"48","DOI":"10.3141\/2080-06","volume":"2080","author":"A Stevanovic","year":"2008","unstructured":"Stevanovic A, Martin P (2008) Split-cycle offset optimization technique and coordinated actuated traffic control evaluated through microsimulation. Transp Res Rec 2080:48\u201356","journal-title":"Transp Res Rec"},{"key":"11530_CR97","doi-asserted-by":"crossref","unstructured":"Sun Q, Zha R, Zhang L, Zhou J, Mei Y, Li Z, Xiong H (2024) Crosslight: Offline-to-online reinforcement learning for cross-city traffic signal control. In: Proceedings of the 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining, pp. 2765\u20132774","DOI":"10.1145\/3637528.3671927"},{"issue":"5","key":"11530_CR98","doi-asserted-by":"publisher","first-page":"1054","DOI":"10.1109\/TNN.1998.712192","volume":"9","author":"RS Sutton","year":"1998","unstructured":"Sutton RS, Barto AG (1998) Reinforcement learning: An introduction. IEEE Trans Neural Networks 9(5):1054\u20131054. https:\/\/doi.org\/10.1109\/TNN.1998.712192","journal-title":"IEEE Trans Neural Networks"},{"key":"11530_CR99","doi-asserted-by":"publisher","first-page":"8140","DOI":"10.1109\/ACCESS.2024.3525119","volume":"13","author":"YH Taher","year":"2025","unstructured":"Taher YH, Mandeep JS, Islam MT, Abdulhae OT, Shakir AT, Islam MS, Soliman MS (2025) Filter for traffic congestion prediction: Leveraging traffic control signal actions for dynamic state estimation. IEEE Access 13:8140\u20138157. https:\/\/doi.org\/10.1109\/ACCESS.2024.3525119","journal-title":"IEEE Access"},{"issue":"6","key":"11530_CR100","doi-asserted-by":"publisher","first-page":"2687","DOI":"10.1109\/TCYB.2019.2904742","volume":"50","author":"T Tan","year":"2019","unstructured":"Tan T, Bao F, Deng Y, Jin A, Dai Q, Wang J (2019) Cooperative deep reinforcement learning for large-scale traffic grid signal control. IEEE transactions on cybernetics 50(6):2687\u20132700","journal-title":"IEEE transactions on cybernetics"},{"issue":"03","key":"11530_CR101","doi-asserted-by":"publisher","first-page":"471","DOI":"10.1142\/S0219525911003104","volume":"14","author":"ME Taylor","year":"2011","unstructured":"Taylor ME, Jain M, Tandon P, Yokoo M, Tambe M (2011) Distributed on-line multi-agent optimization under uncertainty: Balancing exploration and exploitation. Adv Complex Syst 14(03):471\u2013528","journal-title":"Adv Complex Syst"},{"key":"11530_CR102","volume-title":"Traffic light control using sarsa with three state representations","author":"TL Thorpe","year":"1996","unstructured":"Thorpe TL, Anderson CW (1996) Traffic light control using sarsa with three state representations. Technical report, Citeseer"},{"key":"11530_CR103","doi-asserted-by":"publisher","first-page":"513","DOI":"10.1016\/j.procs.2017.05.327","volume":"109","author":"S Touhbi","year":"2017","unstructured":"Touhbi S, Babram MA, Nguyen-Huu T, Marilleau N, Hbid ML, Cambier C, Stinckwich S (2017) Adaptive traffic signal control: Exploring reward definition for reinforcement learning. Procedia Computer Science 109:513\u2013520","journal-title":"Procedia Computer Science"},{"key":"11530_CR104","doi-asserted-by":"crossref","unstructured":"Varaiya P (2013) The max-pressure controller for arbitrary networks of signalized intersections. In: Advances in Dynamic Network Modeling in Complex Transportation Systems, pp. 27\u201366. Springer, ???","DOI":"10.1007\/978-1-4614-6243-9_2"},{"issue":"9","key":"11530_CR105","doi-asserted-by":"publisher","first-page":"1005","DOI":"10.1049\/iet-its.2018.5170","volume":"12","author":"C-H Wan","year":"2018","unstructured":"Wan C-H, Hwang M-C (2018) Value-based deep reinforcement learning for adaptive isolated intersection signal control. IET Intel Transport Syst 12(9):1005\u20131010","journal-title":"IET Intel Transport Syst"},{"issue":"7","key":"11530_CR106","doi-asserted-by":"publisher","first-page":"6774","DOI":"10.1109\/TITS.2021.3062072","volume":"23","author":"M Wang","year":"2021","unstructured":"Wang M, Wu L, Li J, He L (2021) Traffic signal control with reinforcement learning based on region-aware cooperative strategy. IEEE Trans Intell Transp Syst 23(7):6774\u20136785","journal-title":"IEEE Trans Intell Transp Syst"},{"issue":"2","key":"11530_CR107","doi-asserted-by":"publisher","first-page":"12","DOI":"10.1145\/3447556.3447565","volume":"22","author":"H Wei","year":"2021","unstructured":"Wei H, Zheng G, Gayah V, Li Z (2021) Recent advances in reinforcement learning for traffic signal control: A survey of models and evaluation. ACM SIGKDD Explorations Newsl 22(2):12\u201318","journal-title":"ACM SIGKDD Explorations Newsl"},{"key":"11530_CR108","doi-asserted-by":"crossref","unstructured":"Wei H, Chen C, Zheng G, Wu K, Gayah V, Xu K, Li Z (2019) Presslight: Learning max pressure control to coordinate traffic signals in arterial network. In: Proceedings of the 25th ACM SIGKDD International Conference on Knowledge Discovery & Data Mining, pp. 1290\u20131298","DOI":"10.1145\/3292500.3330949"},{"key":"11530_CR109","doi-asserted-by":"crossref","unstructured":"Wei H, Xu N, Zhang H, Zheng G, Zang X, Chen C, Zhang W, Zhu Y, Xu K, Li Z (2019) Colight: Learning network-level cooperation for traffic signal control. In: Proceedings of the 28th ACM International Conference on Information and Knowledge Management, pp. 1913\u20131922","DOI":"10.1145\/3357384.3357902"},{"key":"11530_CR110","doi-asserted-by":"crossref","unstructured":"Wei H, Zheng G, Yao H, Li Z (2018) Intellilight: A reinforcement learning approach for intelligent traffic light control. In: Proceedings of the 24th ACM SIGKDD International Conference on Knowledge Discovery & Data Mining, pp. 2496\u20132505","DOI":"10.1145\/3219819.3220096"},{"key":"11530_CR111","doi-asserted-by":"crossref","unstructured":"Wen K, Qu S, Zhang Y (2008) A Stochastic Adaptive Control Model for Isolated Intersections. In: 2007 IEEE International Conference on Robotics and Biomimetics, Sanya, China, pp. 2256\u20132260. IEEE","DOI":"10.1109\/ROBIO.2007.4522521"},{"key":"11530_CR112","unstructured":"Wiering MA, et al (2000) Multi-agent reinforcement learning for traffic light control. In: Machine Learning: Proceedings of the Seventeenth International Conference (ICML\u20192000), pp. 1151\u20131158"},{"key":"11530_CR113","doi-asserted-by":"crossref","unstructured":"Wu Q, Li M, Shen J, L\u00fc L, Du B, Zhang K (2023) Transformerlight: A novel sequence modeling based traffic signaling mechanism via gated transformer. In: Proceedings of the 29th ACM SIGKDD Conference on Knowledge Discovery and Data Mining, pp. 2639\u20132647","DOI":"10.1145\/3580305.3599530"},{"key":"11530_CR114","doi-asserted-by":"crossref","unstructured":"Wu L, Wang M, Wu D, Wu J (2021) Dynstgat: Dynamic spatial-temporal graph attention network for traffic signal control. In: Proceedings of the 30th ACM International Conference on Information & Knowledge Management, pp. 2150\u20132159","DOI":"10.1145\/3459637.3482254"},{"key":"11530_CR115","doi-asserted-by":"crossref","unstructured":"Xiao H, Li H, Qi S, Zhang J, Cai D (2025) Fglight: Learning neighbor-level information for traffic signal control. In: Proceedings of the 24th International Conference on Autonomous Agents and Multiagent Systems. AAMAS \u201925, pp. 2181\u20132189. International Foundation for Autonomous Agents and Multiagent Systems, Richland, SC","DOI":"10.65109\/BWFG7387"},{"key":"11530_CR116","doi-asserted-by":"crossref","unstructured":"Xing D, Zheng Q, Liu Q, Pan G (2022) Tinylight: Adaptive traffic signal control on devices with extremely limited resources. arXiv preprint arXiv:2205.00427","DOI":"10.24963\/ijcai.2022\/555"},{"issue":"1","key":"11530_CR117","volume":"2013","author":"L-H Xu","year":"2013","unstructured":"Xu L-H, Xia X-H, Luo Q (2013) The study of reinforcement learning for traffic self-adaptive control under multiagent markov game environment. Math Probl Eng 2013(1):962869","journal-title":"Math Probl Eng"},{"issue":"1","key":"11530_CR118","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1080\/15472450.2018.1527694","volume":"24","author":"M Xu","year":"2020","unstructured":"Xu M, Wu J, Huang L, Zhou R, Wang T, Hu D (2020) Network-wide traffic signal control based on the discovery of critical nodes and deep reinforcement learning. Journal of Intelligent Transportation Systems 24(1):1\u201310","journal-title":"Journal of Intelligent Transportation Systems"},{"issue":"7","key":"11530_CR119","doi-asserted-by":"publisher","first-page":"7552","DOI":"10.1109\/TITS.2022.3156816","volume":"24","author":"K Xu","year":"2022","unstructured":"Xu K, Huang J, Kong L, Yu J, Chen G (2022) Pv-tsc: Learning to control traffic signals for pedestrian and vehicle traffic in 6g era. IEEE Trans Intell Transp Syst 24(7):7552\u20137563","journal-title":"IEEE Trans Intell Transp Syst"},{"key":"11530_CR120","doi-asserted-by":"crossref","unstructured":"Xu B, Wang Y, Wang Z, Jia H, Lu Z (2021) Hierarchically and cooperatively learning traffic signal control. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, pp. 669\u2013677","DOI":"10.1609\/aaai.v35i1.16147"},{"key":"11530_CR121","doi-asserted-by":"crossref","unstructured":"Xu N, Zheng G, Xu K, Zhu Y, Li Z (2019) Targeted knowledge transfer for learning traffic signal plans. In: Advances in Knowledge Discovery and Data Mining: 23rd Pacific-Asia Conference, PAKDD 2019, Macau, China, April 14\u201317, 2019, Proceedings, Part II 23, pp. 175\u2013187. Springer","DOI":"10.1007\/978-3-030-16145-3_14"},{"key":"11530_CR122","doi-asserted-by":"crossref","unstructured":"Yang T, Fan W (2024) Enhancing robustness of deep reinforcement learning based adaptive traffic signal controllers in mixed traffic environments through data fusion and multi-discrete actions. IEEE Transactions on Intelligent Transportation Systems","DOI":"10.1109\/TITS.2024.3399066"},{"key":"11530_CR123","doi-asserted-by":"crossref","unstructured":"Yang Q, Xie Z, Wei H, Zhang D, Yang Y (2024) Mallight: Influence-aware coordinated traffic signal control for traffic signal malfunctions. In: Proceedings of the 33rd ACM International Conference on Information and Knowledge Management, pp. 2879\u20132889","DOI":"10.1145\/3627673.3679605"},{"issue":"3","key":"11530_CR124","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3068287","volume":"50","author":"K-LA Yau","year":"2017","unstructured":"Yau K-LA, Qadir J, Khoo HL, Ling MH, Komisarczuk P (2017) A survey on reinforcement learning models and algorithms for traffic signal control. ACM Computing Surveys (CSUR) 50(3):1\u201338","journal-title":"ACM Computing Surveys (CSUR)"},{"key":"11530_CR125","doi-asserted-by":"publisher","DOI":"10.1016\/j.trc.2022.103991","volume":"149","author":"M Yazdani","year":"2023","unstructured":"Yazdani M, Sarvi M, Bagloee SA, Nassir N, Price J, Parineh H (2023) Intelligent vehicle pedestrian light (ivpl): A deep reinforcement learning approach for traffic signal control. Transportation research part C emerging technologies 149:103991","journal-title":"Transportation research part C emerging technologies"},{"key":"11530_CR126","doi-asserted-by":"crossref","unstructured":"Ye Y, Zhou Y, Ding J, Wang T, Chen M, Lian X (2023) Initlight: initial model generation for traffic signal control using adversarial inverse reinforcement learning. In: IJCAI","DOI":"10.24963\/ijcai.2023\/550"},{"key":"11530_CR127","doi-asserted-by":"crossref","unstructured":"Yi C, Wu J, Ren Y, Ran Y, Lou Y (2022) A spatial-temporal deep reinforcement learning model for large-scale centralized traffic signal control. In: 2022 IEEE 25th International Conference on Intelligent Transportation Systems (ITSC), pp. 275\u2013280. IEEE","DOI":"10.1109\/ITSC55140.2022.9922459"},{"key":"11530_CR128","unstructured":"Yuan Z, Lai S, Liu H (2025) CoLLMLight: Cooperative Large Language Model Agents for Network-Wide Traffic Signal Control . https:\/\/arxiv.org\/abs\/2503.11739"},{"issue":"1","key":"11530_CR129","doi-asserted-by":"publisher","first-page":"404","DOI":"10.1109\/TITS.2019.2958859","volume":"22","author":"R Zhang","year":"2020","unstructured":"Zhang R, Ishikawa A, Wang W, Striner B, Tonguz OK (2020) Using reinforcement learning with partial vehicle detection for intelligent traffic signal control. IEEE Trans Intell Transp Syst 22(1):404\u2013415","journal-title":"IEEE Trans Intell Transp Syst"},{"issue":"12","key":"11530_CR130","doi-asserted-by":"publisher","first-page":"25157","DOI":"10.1109\/TITS.2022.3173490","volume":"23","author":"C Zhang","year":"2022","unstructured":"Zhang C, Tian Y, Zhang Z, Xue W, Xie X, Yang T, Ge X, Chen R (2022) Neighborhood cooperative multiagent reinforcement learning for adaptive traffic signal control in epidemic regions. IEEE Trans Intell Transp Syst 23(12):25157\u201325168","journal-title":"IEEE Trans Intell Transp Syst"},{"issue":"1","key":"11530_CR131","doi-asserted-by":"publisher","first-page":"178","DOI":"10.1109\/TITS.2022.3216203","volume":"24","author":"W Zhang","year":"2022","unstructured":"Zhang W, Yan C, Li X, Fang L, Wu Y-J, Li J (2022) Distributed signal control of arterial corridors using multi-agent deep reinforcement learning. IEEE Trans Intell Transp Syst 24(1):178\u2013190","journal-title":"IEEE Trans Intell Transp Syst"},{"key":"11530_CR132","doi-asserted-by":"crossref","unstructured":"Zheng G, Xiong Y, Zang X, Feng J, Wei H, Zhang H, Li Y, Xu K, Li Z (2019) Learning phase competition for traffic signal control. In: Proceedings of the 28th ACM International Conference on Information and Knowledge Management, pp. 1963\u20131972","DOI":"10.1145\/3357384.3357900"},{"key":"11530_CR133","doi-asserted-by":"crossref","unstructured":"Zhou P, Braud T, Alhilal A, Hui P, Kangasharju J (2019) Erl: Edge based reinforcement learning for optimized urban traffic light control. In: 2019 IEEE International Conference on Pervasive Computing and Communications Workshops (PerCom Workshops), pp. 849\u2013854. IEEE","DOI":"10.1109\/PERCOMW.2019.8730706"},{"key":"11530_CR134","doi-asserted-by":"crossref","unstructured":"Zhu H, Sun F, Tang K, Han T, Xiang J (2024) A coordination graph based framework for network traffic signal control. IEEE Transactions on Intelligent Transportation Systems","DOI":"10.1109\/TITS.2024.3405171"}],"container-title":["Artificial Intelligence Review"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10462-026-11530-9","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-026-11530-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-026-11530-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T23:14:48Z","timestamp":1778109288000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10462-026-11530-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,9]]},"references-count":134,"journal-issue":{"issue":"5","published-online":{"date-parts":[[2026,5]]}},"alternative-id":["11530"],"URL":"https:\/\/doi.org\/10.1007\/s10462-026-11530-9","relation":{},"ISSN":["1573-7462"],"issn-type":[{"value":"1573-7462","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3,9]]},"assertion":[{"value":"24 June 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 February 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 March 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no competing interests to declare that arerelevant to the content of this article","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interest"}}],"article-number":"128"}}