{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,19]],"date-time":"2025-12-19T01:08:58Z","timestamp":1766106538239,"version":"3.48.0"},"reference-count":46,"publisher":"Springer Science and Business Media LLC","issue":"17","license":[{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100018537","name":"National Science and Technology Major Project","doi-asserted-by":"publisher","award":["2021ZD0112702"],"award-info":[{"award-number":["2021ZD0112702"]}],"id":[{"id":"10.13039\/501100018537","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62373100"],"award-info":[{"award-number":["62373100"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2025,11]]},"DOI":"10.1007\/s10489-025-07020-0","type":"journal-article","created":{"date-parts":[[2025,11,29]],"date-time":"2025-11-29T03:18:24Z","timestamp":1764386304000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["A deep reinforcement learning model for traffic signal control with multi-expert participating in exploration"],"prefix":"10.1007","volume":"55","author":[{"given":"Hui","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhicheng","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shiyi","family":"Gu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7555-162X","authenticated-orcid":false,"given":"Ya","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,11,29]]},"reference":[{"issue":"2","key":"7020_CR1","doi-asserted-by":"publisher","first-page":"1187","DOI":"10.1109\/TVT.2021.3069921","volume":"71","author":"A Boukerche","year":"2021","unstructured":"Boukerche A, Zhong D, Sun P (2021) A novel reinforcement learning-based cooperative traffic signal system through max-pressure control. IEEE Trans Veh Technol 71(2):1187\u20131198","journal-title":"IEEE Trans Veh Technol"},{"key":"7020_CR2","doi-asserted-by":"crossref","unstructured":"Chen C, Wei H, Xu N, et\u00a0al (2020) Toward a thousand lights: Decentralized deep reinforcement learning for large-scale traffic signal control. In: Proceedings of the AAAI conference on artificial intelligence, pp 3414\u20133421","DOI":"10.1609\/aaai.v34i04.5744"},{"key":"7020_CR3","doi-asserted-by":"crossref","unstructured":"Chergui O, Sayad L (2023) Mitigating congestion in multi-agent traffic signal control: an efficient self-attention proximal policy optimization approach. International Journal of Information Technology pp 1\u201310","DOI":"10.1007\/s41870-023-01545-8"},{"key":"7020_CR4","unstructured":"Christiano PF, Leike J, Brown T, et\u00a0al (2017) Deep reinforcement learning from human preferences. Advances in neural information processing systems 30"},{"key":"7020_CR5","unstructured":"Chu T, Wang J, Codeca L, et\u00a0al (2019) Multi-agent deep reinforcement learning for large-scale traffic signal control. IEEE Transactions on Intelligent Transportation Systems pp 1\u201310"},{"key":"7020_CR6","doi-asserted-by":"publisher","unstructured":"Cools SB, Gershenson C, D\u2019Hooghe B (2013) Self-organizing traffic lights: A realistic simulation. Advances in applied self-organizing systems pp 45\u201355. https:\/\/doi.org\/10.1109\/JAS.2023.123561","DOI":"10.1109\/JAS.2023.123561"},{"issue":"7","key":"7020_CR7","doi-asserted-by":"publisher","first-page":"7496","DOI":"10.1109\/TITS.2021.3070835","volume":"23","author":"FX Devailly","year":"2021","unstructured":"Devailly FX, Larocque D, Charlin L (2021) Ig-rl: Inductive graph reinforcement learning for massive-scale traffic signal control. IEEE Trans Intell Transp Syst 23(7):7496\u20137507","journal-title":"IEEE Trans Intell Transp Syst"},{"issue":"2","key":"7020_CR8","doi-asserted-by":"publisher","first-page":"511","DOI":"10.1109\/TFUZZ.2022.3214001","volume":"31","author":"B Fang","year":"2022","unstructured":"Fang B, Zheng C, Wang H et al (2022) Two-stream fused fuzzy deep neural network for multiagent learning. IEEE Trans Fuzzy Syst 31(2):511\u2013520","journal-title":"IEEE Trans Fuzzy Syst"},{"issue":"3","key":"7020_CR9","doi-asserted-by":"publisher","first-page":"04024001","DOI":"10.1061\/JTEPBS.TEENG-8261","volume":"150","author":"G Han","year":"2024","unstructured":"Han G, Liu X, Wang H et al (2024) An attention reinforcement learning-based strategy for large-scale adaptive traffic signal control system. Journal of Transportation Engineering Part A Systems 150(3):04024001","journal-title":"Journal of Transportation Engineering Part A Systems"},{"key":"7020_CR10","doi-asserted-by":"crossref","unstructured":"Hester T, Vecerik M, Pietquin O, et\u00a0al (2018) Deep q-learning from demonstrations. In: Proceedings of the AAAI conference on artificial intelligence","DOI":"10.1609\/aaai.v32i1.11757"},{"issue":"9","key":"7020_CR11","doi-asserted-by":"publisher","first-page":"1715","DOI":"10.1049\/itr2.12364","volume":"17","author":"T Hu","year":"2023","unstructured":"Hu T, Hu Z, Lu Z et al (2023) Dynamic traffic signal control using mean field multi-agent reinforcement learning in large scale road-networks. IET Intel Transport Syst 17(9):1715\u20131728","journal-title":"IET Intel Transport Syst"},{"key":"7020_CR12","unstructured":"Hunt P, Robertson D, Bretherton R, et\u00a0al (1982) The scoot on-line traffic signal optimisation technique. Traffic Engineering & Control 23(4)"},{"key":"7020_CR13","unstructured":"Ibarz B, Leike J, Pohlen T, et\u00a0al (2018) Reward learning from human preferences and demonstrations in atari. Advances in neural information processing systems 31"},{"key":"7020_CR14","doi-asserted-by":"crossref","unstructured":"Jayawardana V, Landler A, Wu C (2021) Mixed autonomous supervision in traffic signal control. In: 2021 IEEE International Intelligent Transportation Systems Conference (ITSC), IEEE, pp 1767\u20131773","DOI":"10.1109\/ITSC48978.2021.9565053"},{"key":"7020_CR15","volume-title":"Traffic signal timing manual","author":"P Koonce","year":"2008","unstructured":"Koonce P, Rodegerdts L (2008) Traffic signal timing manual. Tech. rep, Federal Highway Administration, United States"},{"key":"7020_CR16","unstructured":"Li Q, Peng Z, Zhou B (2022) Efficient learning of safe driving policy via human-ai copilot optimization. arXiv:2202.10341"},{"key":"7020_CR17","doi-asserted-by":"crossref","unstructured":"Li X, Fang J, Du K, et\u00a0al (2023) Uav obstacle avoidance by human-in-the-loop reinforcement in arbitrary 3d environment. In: 2023 42nd Chinese Control Conference (CCC), IEEE, pp 3589\u20133595","DOI":"10.23919\/CCC58697.2023.10240962"},{"issue":"9","key":"7020_CR18","doi-asserted-by":"publisher","first-page":"1987","DOI":"10.1109\/JAS.2024.124365","volume":"11","author":"Y Li","year":"2024","unstructured":"Li Y, Zhang Y, Li X et al (2024) Regional multi-agent cooperative reinforcement learning for city-level traffic grid signal control. IEEE\/CAA Journal of Automatica Sinica 11(9):1987\u20131998","journal-title":"IEEE\/CAA Journal of Automatica Sinica"},{"key":"7020_CR19","doi-asserted-by":"crossref","unstructured":"Lin WY, Song YZ, Ruan BK, et\u00a0al (2023) Temporal difference-aware graph convolutional reinforcement learning for multi-intersection traffic signal control. IEEE Transactions on Intelligent Transportation Systems","DOI":"10.1109\/TITS.2023.3311426"},{"key":"7020_CR20","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2023.120458","volume":"228","author":"J Liu","year":"2023","unstructured":"Liu J, Qin S, Su M et al (2023) Traffic signal control using reinforcement learning based on the teacher-student framework. Expert Syst Appl 228:120458","journal-title":"Expert Syst Appl"},{"key":"7020_CR21","volume-title":"Scats-a traffic responsive method of controlling urban traffic","author":"P Lowrie","year":"1990","unstructured":"Lowrie P (1990) Scats-a traffic responsive method of controlling urban traffic. Sales information brochure published by Roads & Traffic Authority, Sydney, Australia"},{"key":"7020_CR22","doi-asserted-by":"crossref","unstructured":"Luo B, Wu Z, Zhou F, et\u00a0al (2023) Human-in-the-loop reinforcement learning in continuous-action space. IEEE Transactions on Neural Networks and Learning Systems","DOI":"10.1109\/TNNLS.2023.3289315"},{"key":"7020_CR23","unstructured":"Metcalf K, Sarabia M, Theobald BJ (2022) Rewards encoding environment dynamics improves preference-based reinforcement learning. arXiv:2211.06527"},{"issue":"4","key":"7020_CR24","doi-asserted-by":"publisher","first-page":"877","DOI":"10.1109\/JAS.2023.123561","volume":"10","author":"Q Miao","year":"2023","unstructured":"Miao Q, Zheng W, Lv Y et al (2023) Dao to hanoi via desci: Ai paradigm shifts from alphago to chatgpt. IEEE\/CAA Journal of Automatica Sinica 10(4):877\u2013897. https:\/\/doi.org\/10.1109\/JAS.2023.123561","journal-title":"IEEE\/CAA Journal of Automatica Sinica"},{"key":"7020_CR25","doi-asserted-by":"publisher","unstructured":"Nguyen DVA, Azevedo CL, Toledo T, et\u00a0al (2025) Robustness of reinforcement learning-based traffic signal control under incidents: A comparative study. arXiv e-prints https:\/\/doi.org\/10.48550\/arXiv.2506.13836","DOI":"10.48550\/arXiv.2506.13836"},{"key":"7020_CR26","unstructured":"Peng Z, Li Q, Liu C, et\u00a0al (2022) Safe driving via expert guided policy optimization. In: Conference on Robot Learning, PMLR, pp 1554\u20131563"},{"key":"7020_CR27","doi-asserted-by":"crossref","unstructured":"Piot B, Geist M, Pietquin O (2014) Boosted bellman residual minimization handling expert demonstrations. In: Machine Learning and Knowledge Discovery in Databases, Springer, pp 549\u2013564","DOI":"10.1007\/978-3-662-44851-9_35"},{"key":"7020_CR28","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2023.121111","volume":"235","author":"F Ren","year":"2024","unstructured":"Ren F, Dong W, Zhao X et al (2024) Two-layer coordinated reinforcement learning for traffic signal control in traffic network. Expert Syst Appl 235:121111","journal-title":"Expert Syst Appl"},{"key":"7020_CR29","doi-asserted-by":"crossref","unstructured":"Silver D, Huang A, Maddison CJ, et\u00a0al (2016) Mastering the game of go with deep neural networks and tree search. nature 529(7587):484\u2013489","DOI":"10.1038\/nature16961"},{"key":"7020_CR30","doi-asserted-by":"crossref","unstructured":"Varaiya P (2013) The max-pressure controller for arbitrary networks of signalized intersections. In: Advances in dynamic network modeling in complex transportation systems. Springer, p 27\u201366","DOI":"10.1007\/978-1-4614-6243-9_2"},{"key":"7020_CR31","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2022.109166","volume":"250","author":"M Wang","year":"2022","unstructured":"Wang M, Wu L, Li M et al (2022) Meta-learning based spatial-temporal graph attention network for traffic signal control. Knowl-Based Syst 250:109166","journal-title":"Knowl-Based Syst"},{"key":"7020_CR32","unstructured":"Wang W, Qiao T, Ma J, et\u00a0al (2023) Real-time network-level traffic signal control: An explicit multiagent coordination method. arXiv:2306.08843"},{"key":"7020_CR33","doi-asserted-by":"crossref","unstructured":"Wei H, Xu N, Zhang H, et\u00a0al (2019) Colight: Learning network-level cooperation for traffic signal control. In: Proceedings of the 28th ACM international conference on information and knowledge management, pp 1913\u20131922","DOI":"10.1145\/3357384.3357902"},{"key":"7020_CR34","doi-asserted-by":"crossref","unstructured":"Wu HN, Wang M (2023) Human-in-the-loop behavior modeling via an integral concurrent adaptive inverse reinforcement learning. IEEE Transactions on Neural Networks and Learning Systems","DOI":"10.1109\/TNNLS.2023.3259581"},{"key":"7020_CR35","unstructured":"Wu Q, Zhang L, Shen J, et\u00a0al (2021) Efficient pressure: Improving efficiency for signalized intersections. arXiv:2112.02336"},{"issue":"6","key":"7020_CR36","doi-asserted-by":"publisher","first-page":"6248","DOI":"10.1007\/s10489-022-03208-w","volume":"53","author":"L Yan","year":"2023","unstructured":"Yan L, Zhu L, Song K et al (2023) Graph cooperation deep reinforcement learning for ecological urban traffic signal control. Appl Intell 53(6):6248\u20136265","journal-title":"Appl Intell"},{"key":"7020_CR37","doi-asserted-by":"publisher","first-page":"55","DOI":"10.1016\/j.ins.2023.03.087","volume":"634","author":"S Yang","year":"2023","unstructured":"Yang S (2023) Hierarchical graph multi-agent reinforcement learning for traffic signal control. Inf Sci 634:55\u201372","journal-title":"Inf Sci"},{"key":"7020_CR38","doi-asserted-by":"publisher","first-page":"243","DOI":"10.1016\/j.inffus.2023.02.009","volume":"94","author":"S Yang","year":"2023","unstructured":"Yang S, Yang B, Zeng Z et al (2023) Causal inference multi-agent reinforcement learning for traffic signal control. Information Fusion 94:243\u2013256","journal-title":"Information Fusion"},{"issue":"12","key":"7020_CR39","doi-asserted-by":"publisher","first-page":"25157","DOI":"10.1109\/TITS.2022.3173490","volume":"23","author":"C Zhang","year":"2022","unstructured":"Zhang C, Tian Y, Zhang Z et al (2022a) Neighborhood cooperative multiagent reinforcement learning for adaptive traffic signal control in epidemic regions. IEEE Trans Intell Transp Syst 23(12):25157\u201325168","journal-title":"IEEE Trans Intell Transp Syst"},{"issue":"000","key":"7020_CR40","first-page":"17","volume":"199","author":"G Zhang","year":"2024","unstructured":"Zhang G, Chang F, Jin J et al (2024) Multi-objective deep reinforcement learning approach for adaptive traffic signal control system with concurrent optimization of safety, efficiency, and decarbonization at intersections. Accident Analysis & Prevention 199(000):17","journal-title":"Accident Analysis & Prevention"},{"key":"7020_CR41","doi-asserted-by":"crossref","unstructured":"Zhang H, Feng S, Liu C, et\u00a0al (2019) Cityflow: A multi-agent reinforcement learning environment for large scale city traffic scenario. The World Wide Web Conference","DOI":"10.1145\/3308558.3314139"},{"key":"7020_CR42","unstructured":"Zhang L, Wu Q, Shen J, et\u00a0al (2022b) Expression might be enough: representing pressure and demand for reinforcement learning based traffic signal control. In: International Conference on Machine Learning, PMLR, pp 26645\u201326654"},{"key":"7020_CR43","doi-asserted-by":"crossref","unstructured":"Zhang L, Xie S, Deng J (2023) Leveraging queue length and attention mechanisms for enhanced traffic signal control optimization. In: Joint European Conference on Machine Learning and Knowledge Discovery in Databases, Springer, pp 141\u2013156","DOI":"10.1007\/978-3-031-43430-3_9"},{"issue":"1","key":"7020_CR44","doi-asserted-by":"publisher","first-page":"178","DOI":"10.1109\/TITS.2022.3216203","volume":"24","author":"W Zhang","year":"2022","unstructured":"Zhang W, Yan C, Li X et al (2022c) Distributed signal control of arterial corridors using multi-agent deep reinforcement learning. IEEE Trans Intell Transp Syst 24(1):178\u2013190","journal-title":"IEEE Trans Intell Transp Syst"},{"key":"7020_CR45","doi-asserted-by":"crossref","unstructured":"Zhou Z, Zhang Y, Li X (2024) A deep reinforcement learning model for large-scale traffic signal control based on graph meta learning using local subgraphs. SCIENCE CHINA Information Sciences under review","DOI":"10.1007\/s11432-023-4280-6"},{"key":"7020_CR46","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2023.110696","volume":"275","author":"R Zhu","year":"2023","unstructured":"Zhu R, Ding W, Wu S et al (2023) Auto-learning communication reinforcement learning for multi-intersection traffic light control. Knowl-Based Syst 275:110696","journal-title":"Knowl-Based Syst"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-025-07020-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-025-07020-0","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-025-07020-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,19]],"date-time":"2025-12-19T01:04:35Z","timestamp":1766106275000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-025-07020-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11]]},"references-count":46,"journal-issue":{"issue":"17","published-print":{"date-parts":[[2025,11]]}},"alternative-id":["7020"],"URL":"https:\/\/doi.org\/10.1007\/s10489-025-07020-0","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"type":"print","value":"0924-669X"},{"type":"electronic","value":"1573-7497"}],"subject":[],"published":{"date-parts":[[2025,11]]},"assertion":[{"value":"7 October 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 November 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 November 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of Interest"}}],"article-number":"1125"}}