{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T09:08:27Z","timestamp":1784538507209,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":31,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,4,13]],"date-time":"2026-04-13T00:00:00Z","timestamp":1776038400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"Taighde Eireann \u2013 Research Ireland","award":["21\/FFP-A\/8957"],"award-info":[{"award-number":["21\/FFP-A\/8957"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,13]]},"DOI":"10.1145\/3788550.3794864","type":"proceedings-article","created":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T08:46:11Z","timestamp":1784537171000},"page":"30-35","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Balancing Multiple Objectives in Urban Traffic Control with Reinforcement Learning from AI Feedback"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3739-8409","authenticated-orcid":false,"given":"Chenyang","family":"Zhao","sequence":"first","affiliation":[{"name":"School of Computer Science and Statistics, Trinity College Dublin, Dublin, Ireland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0822-3490","authenticated-orcid":false,"given":"Vinny","family":"Cahill","sequence":"additional","affiliation":[{"name":"School of Computer Science and Statistics, Trinity College Dublin, Dublin, Ireland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0621-5400","authenticated-orcid":false,"given":"Ivana","family":"Dusparic","sequence":"additional","affiliation":[{"name":"School of Computer Science and Statistics, Trinity College Dublin, Dublin, Ireland"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,20]]},"reference":[{"key":"e_1_3_3_2_2_2","unstructured":"Lucas\u00a0N. Alegre. 2019. SUMO-RL. https:\/\/github.com\/LucasAlegre\/sumo-rl."},{"key":"e_1_3_3_2_3_2","doi-asserted-by":"crossref","unstructured":"Marc\u00a0G Bellemare Salvatore Candido Pablo\u00a0Samuel Castro Jun Gong Marlos\u00a0C Machado Subhodeep Moitra Sameera\u00a0S Ponda and Ziyu Wang. 2020. Autonomous navigation of stratospheric balloons using reinforcement learning. Nature 588 7836 (2020) 77\u201382.","DOI":"10.1038\/s41586-020-2939-8"},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"crossref","unstructured":"Ralph\u00a0Allan Bradley and Milton\u00a0E Terry. 1952. Rank analysis of incomplete block designs: I. the method of paired comparisons. Biometrika 39 3\/4 (1952) 324\u2013345.","DOI":"10.1093\/biomet\/39.3-4.324"},{"key":"e_1_3_3_2_5_2","unstructured":"Alexander Bukharin Yixiao Li Pengcheng He and Tuo Zhao. 2023. Deep Reinforcement Learning from Hierarchical Preference Design. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2309.02632 (2023)."},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"crossref","first-page":"214","DOI":"10.1145\/3643915.3644089","volume-title":"Proceedings of the 19th International Symposium on Software Engineering for Adaptive and Self-Managing Systems","author":"Chan Kenneth\u00a0H","year":"2024","unstructured":"Kenneth\u00a0H Chan, Sol Zilberman, Nick Polanco, Joshua\u00a0E Siegel, and Betty\u00a0HC Cheng. 2024. SafeDriveRL: Combining non-cooperative game theory with reinforcement learning to explore and mitigate human-based uncertainty for autonomous vehicles. In Proceedings of the 19th International Symposium on Software Engineering for Adaptive and Self-Managing Systems. 214\u2013220."},{"key":"e_1_3_3_2_7_2","unstructured":"Paul\u00a0F Christiano Jan Leike Tom Brown Miljan Martic Shane Legg and Dario Amodei. 2017. Deep reinforcement learning from human preferences. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_3_2_8_2","unstructured":"Juntao Gao Yulong Shen Jia Liu Minoru Ito and Norio Shiratori. 2017. Adaptive traffic signal control: Deep reinforcement learning algorithm with experience replay and target network. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1705.02755 (2017)."},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"crossref","unstructured":"Shuaiyi Huang Mara Levy Anubhav Gupta Daniel Ekpo Ruijie Zheng and Abhinav Shrivastava. 2025. Trend: Tri-teaching for robust preference-based reinforcement learning with demonstrations. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2505.06079 (2025).","DOI":"10.1109\/ICRA55743.2025.11128278"},{"key":"e_1_3_3_2_10_2","unstructured":"Martin Klissarov Pierluca D\u2019Oro Shagun Sodhani Roberta Raileanu Pierre-Luc Bacon Pascal Vincent Amy Zhang and Mikael Henaff. 2023. Motif: Intrinsic motivation from artificial intelligence feedback. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.00166 (2023)."},{"key":"e_1_3_3_2_11_2","unstructured":"Minae Kwon Sang\u00a0Michael Xie Kalesha Bullard and Dorsa Sadigh. 2023. Reward design with language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.00001 (2023)."},{"key":"e_1_3_3_2_12_2","unstructured":"Kimin Lee Laura Smith and Pieter Abbeel. 2021. Pebble: Feedback-efficient interactive reinforcement learning via relabeling experience and unsupervised pre-training. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2106.05091 (2021)."},{"key":"e_1_3_3_2_13_2","unstructured":"Kimin Lee Laura Smith Anca Dragan and Pieter Abbeel. 2021. B-pref: Benchmarking preference-based reinforcement learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2111.03026 (2021)."},{"key":"e_1_3_3_2_14_2","volume-title":"The 21st IEEE International Conference on Intelligent Transportation SystemsIEEE Intelligent Transportation Systems Conference (ITSC)","author":"Lopez Pablo\u00a0Alvarez","year":"2018","unstructured":"Pablo\u00a0Alvarez Lopez, Michael Behrisch, Laura Bieker-Walz, Jakob Erdmann, Yun-Pang Fl\u00f6tter\u00f6d, Robert Hilbrich, Leonhard L\u00fccken, Johannes Rummel, Peter Wagner, and Evamarie Wie\u00dfner. 2018. Microscopic Traffic Simulation using SUMO, In The 21st IEEE International Conference on Intelligent Transportation Systems. IEEE Intelligent Transportation Systems Conference (ITSC). https:\/\/elib.dlr.de\/124092\/"},{"key":"e_1_3_3_2_15_2","unstructured":"Ilya Loshchilov and Frank Hutter. 2017. Decoupled weight decay regularization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1711.05101 (2017)."},{"key":"e_1_3_3_2_16_2","unstructured":"Tung\u00a0Minh Luu Younghwan Lee Donghoon Lee Sunho Kim Min\u00a0Jun Kim and Chang\u00a0D Yoo. 2025. Enhancing Rating-Based Reinforcement Learning to Effectively Leverage Feedback from Large Vision-Language Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2506.12822 (2025)."},{"key":"e_1_3_3_2_17_2","unstructured":"Yecheng\u00a0Jason Ma William Liang Guanzhi Wang De-An Huang Osbert Bastani Dinesh Jayaraman Yuke Zhu Linxi Fan and Anima Anandkumar. 2023. Eureka: Human-level reward design via coding large language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.12931 (2023)."},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"crossref","unstructured":"Ni Mu Yao Luan and Qing-Shan Jia. 2025. Preference-based multi-objective reinforcement learning. IEEE Transactions on Automation Science and Engineering (2025).","DOI":"10.1109\/TASE.2025.3589271"},{"key":"e_1_3_3_2_19_2","volume-title":"GPT-4.1-nano","year":"2025","unstructured":"OpenAI. 2025. GPT-4.1-nano. OpenAI. https:\/\/platform.openai.com\/docs\/models\/gpt-4.1-nano Large language model accessed via the OpenAI API. Model: gpt-4.1-nano. Accessed: 2025-10-23."},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"crossref","unstructured":"Takumi Saiki and Sachiyo Arai. 2023. Flexible traffic signal control via multi-objective reinforcement learning. IEEE Access 11 (2023) 75875\u201375883.","DOI":"10.1109\/ACCESS.2023.3296537"},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"crossref","unstructured":"Ruiqi Wang Dezhong Zhao Ziqin Yuan Ike Obi and Byung-Cheol Min. 2025. Prefclm: Enhancing preference-based reinforcement learning with crowdsourced large language models. IEEE Robotics and Automation Letters (2025).","DOI":"10.1109\/LRA.2025.3528663"},{"key":"e_1_3_3_2_22_2","unstructured":"Yufei Wang Zhanyi Sun Jesse Zhang Zhou Xian Erdem Biyik David Held and Zackory Erickson. 2024. Rl-vlm-f: Reinforcement learning from vision language foundation model feedback. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2402.03681 (2024)."},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"crossref","first-page":"1290","DOI":"10.1145\/3292500.3330949","volume-title":"Proceedings of the 25th ACM SIGKDD international conference on knowledge discovery & data mining","author":"Wei Hua","year":"2019","unstructured":"Hua Wei, Chacha Chen, Guanjie Zheng, Kan Wu, Vikash Gayah, Kai Xu, and Zhenhui Li. 2019. Presslight: Learning max pressure control to coordinate traffic signals in arterial network. In Proceedings of the 25th ACM SIGKDD international conference on knowledge discovery & data mining. 1290\u20131298."},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"crossref","unstructured":"Hua Wei Guanjie Zheng Vikash Gayah and Zhenhui Li. 2021. Recent advances in reinforcement learning for traffic signal control: A survey of models and evaluation. ACM SIGKDD explorations newsletter 22 2 (2021) 12\u201318.","DOI":"10.1145\/3447556.3447565"},{"key":"e_1_3_3_2_25_2","unstructured":"Christian Wirth Riad Akrour Gerhard Neumann and Johannes F\u00fcrnkranz. 2017. A survey of preference-based reinforcement learning methods. Journal of Machine Learning Research 18 136 (2017) 1\u201346."},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"crossref","unstructured":"Yue Wu Yewen Fan Paul\u00a0Pu Liang Amos Azaria Yuanzhi Li and Tom\u00a0M Mitchell. 2023. Read and reap the rewards: Learning to play atari with the help of instruction manuals. Advances in Neural Information Processing Systems 36 (2023) 1009\u20131023.","DOI":"10.52202\/075280-0048"},{"key":"e_1_3_3_2_27_2","unstructured":"Tianbao Xie Siheng Zhao Chen\u00a0Henry Wu Yitao Liu Qian Luo Victor Zhong Yanchao Yang and Tao Yu. 2023. Text2reward: Reward shaping with language models for reinforcement learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2309.11489 (2023)."},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"crossref","unstructured":"Gongquan Zhang Fangrong Chang Jieling Jin Fan Yang and Helai Huang. 2024. Multi-objective deep reinforcement learning approach for adaptive traffic signal control system with concurrent optimization of safety efficiency and decarbonization at intersections. Accident Analysis & Prevention 199 (2024) 107451.","DOI":"10.1016\/j.aap.2023.107451"},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"crossref","unstructured":"Yuqi Zhang Yingying Zhou Beilei Wang and Jie Song. 2024. MMD-TSC: An adaptive multi-objective traffic signal control for energy saving with traffic efficiency. Energies 17 19 (2024) 5015.","DOI":"10.3390\/en17195015"},{"key":"e_1_3_3_2_30_2","doi-asserted-by":"crossref","unstructured":"Haiyan Zhao Chengcheng Dong Jian Cao and Qingkui Chen. 2024. A survey on deep reinforcement learning approaches for traffic signal control. Engineering Applications of Artificial Intelligence 133 (2024) 108100.","DOI":"10.1016\/j.engappai.2024.108100"},{"key":"e_1_3_3_2_31_2","unstructured":"Guanjie Zheng Xinshi Zang Nan Xu Hua Wei Zhengyao Yu Vikash Gayah Kai Xu and Zhenhui Li. 2019. Diagnosing reinforcement learning for traffic signal control. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1905.04716 (2019)."},{"key":"e_1_3_3_2_32_2","unstructured":"Zihao Zhou Bin Hu Chenyang Zhao Pu Zhang and Bin Liu. 2023. Large language model as a policy teacher for training reinforcement learning agents. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.13373 (2023)."}],"event":{"name":"SEAMS '26: 21st International Conference on Software Engineering for Adaptive and Self-Managing Systems","location":"Rio de Janeiro Brazil","acronym":"SEAMS '26","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering","IEEE CS"]},"container-title":["Proceedings of the 21st International Conference on Software Engineering for Adaptive and Self-Managing Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3788550.3794864","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T08:48:25Z","timestamp":1784537305000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3788550.3794864"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,13]]},"references-count":31,"alternative-id":["10.1145\/3788550.3794864","10.1145\/3788550"],"URL":"https:\/\/doi.org\/10.1145\/3788550.3794864","relation":{},"subject":[],"published":{"date-parts":[[2026,4,13]]},"assertion":[{"value":"2026-07-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}