{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:56:22Z","timestamp":1765310182409,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":57,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755206","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:26:38Z","timestamp":1761377198000},"page":"1490-1499","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Multimodal Dual Population Evolutionary Reinforcement Learning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-6993-3501","authenticated-orcid":false,"given":"Yao","family":"Zhang","sequence":"first","affiliation":[{"name":"Harbin Engineering University, Harbin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4170-8916","authenticated-orcid":false,"given":"Ping","family":"Huang","sequence":"additional","affiliation":[{"name":"Harbin Engineering University, Harbin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-6793-8310","authenticated-orcid":false,"given":"Rui","family":"Zhang","sequence":"additional","affiliation":[{"name":"Harbin Engineering University, Harbin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2017.2743240"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CEC.2016.7744026"},{"key":"e_1_3_2_1_3_1","volume-title":"An overview of evolutionary algorithms for parameter optimization. Evolutionary computation","author":"B\u00e4ck Thomas","year":"1993","unstructured":"Thomas B\u00e4ck and Hans-Paul Schwefel. 1993. An overview of evolutionary algorithms for parameter optimization. Evolutionary computation, Vol. 1, 1 (1993), 1-23."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.34133\/icomputing.0025"},{"key":"e_1_3_2_1_5_1","volume-title":"Evolution strategies-a comprehensive introduction. Natural computing","author":"Beyer Hans-Georg","year":"2002","unstructured":"Hans-Georg Beyer and Hans-Paul Schwefel. 2002. Evolution strategies-a comprehensive introduction. Natural computing, Vol. 1 (2002), 3-52."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.5728"},{"key":"e_1_3_2_1_7_1","volume-title":"Neuroevolution is a competitive alternative to reinforcement learning for skill discovery. arXiv preprint arXiv:2210.03516","author":"Chalumeau Felix","year":"2022","unstructured":"Felix Chalumeau, Raphael Boige, Bryan Lim, Valentin Mac\u00e9, Maxime Allard, Arthur Flajolet, Antoine Cully, and Thomas Pierrot. 2022. Neuroevolution is a competitive alternative to reinforcement learning for skill discovery. arXiv preprint arXiv:2210.03516 (2022)."},{"volume-title":"Black Box Optimization, Machine Learning, and No-Free Lunch Theorems","author":"Chatzilygeroudis Konstantinos","key":"e_1_3_2_1_8_1","unstructured":"Konstantinos Chatzilygeroudis, Antoine Cully, Vassilis Vassiliades, and Jean-Baptiste Mouret. 2021. Quality-diversity optimization: a novel branch of stochastic optimization. In Black Box Optimization, Machine Learning, and No-Free Lunch Theorems. Springer, 109-135."},{"key":"e_1_3_2_1_9_1","volume-title":"Qd-rl: Efficient mixing of quality and diversity in reinforcement learning. arXiv preprint arXiv:2006.08505","author":"Cideron Geoffrey","year":"2020","unstructured":"Geoffrey Cideron, Thomas Pierrot, Nicolas Perrin, Karim Beguir, and Olivier Sigaud. 2020. Qd-rl: Efficient mixing of quality and diversity in reinforcement learning. arXiv preprint arXiv:2006.08505 (2020), 28-73."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/TEVC.2017.2704781"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10479-005-5724-z"},{"key":"e_1_3_2_1_12_1","volume-title":"Reinforcement learning versus evolutionary computation: A survey on hybrid algorithms. Swarm and evolutionary computation","author":"Drugan Madalina M","year":"2019","unstructured":"Madalina M Drugan. 2019. Reinforcement learning versus evolutionary computation: A survey on hybrid algorithms. Swarm and evolutionary computation, Vol. 44 (2019), 228-246."},{"key":"e_1_3_2_1_13_1","volume-title":"International Conference on Learning Representations.","author":"Eysenbach Benjamin","year":"2019","unstructured":"Benjamin Eysenbach, Abhishek Gupta, Julian Ibarz, and Sergey Levine. 2019. Diversity is All You Need: Learning Skills without a Reward Function. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_14_1","volume-title":"International conference on machine learning. PMLR, 1587-1596","author":"Fujimoto Scott","year":"2018","unstructured":"Scott Fujimoto, Herke Hoof, and David Meger. 2018. Addressing function approximation error in actor-critic methods. In International conference on machine learning. PMLR, 1587-1596."},{"key":"e_1_3_2_1_15_1","unstructured":"Tuomas Haarnoja Aurick Zhou Kristian Hartikainen George Tucker Sehoon Ha Jie Tan Vikash Kumar Henry Zhu Abhishek Gupta Pieter Abbeel et al. 2018. Soft actor-critic algorithms and applications. arXiv preprint arXiv:1812.05905 (2018)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1142\/S0129065714500087"},{"key":"e_1_3_2_1_17_1","volume-title":"Discovering Temporally-Aware Reinforcement Learning Algorithms. In The Twelfth International Conference on Learning Representations.","author":"Jackson Matthew Thomas","year":"2024","unstructured":"Matthew Thomas Jackson, Chris Lu, Louis Kirsch, Robert Tjarko Lange, Shimon Whiteson, and Jakob Nicolaus Foerster. 2024. Discovering Temporally-Aware Reinforcement Learning Algorithms. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2019.8794127"},{"key":"e_1_3_2_1_19_1","volume-title":"Sumit Singh Chauhan, and Vijay Kumar","author":"Katoch Sourabh","year":"2021","unstructured":"Sourabh Katoch, Sumit Singh Chauhan, and Vijay Kumar. 2021. A review on genetic algorithm: past, present, and future. Multimedia tools and applications, Vol. 80 (2021), 8091-8126."},{"key":"e_1_3_2_1_20_1","volume-title":"International conference on machine learning. PMLR, 3341-3350","author":"Khadka Shauharda","year":"2019","unstructured":"Shauharda Khadka, Somdeb Majumdar, Tarek Nassar, Zach Dwiel, Evren Tumer, Santiago Miret, Yinyin Liu, and Kagan Tumer. 2019. Collaborative evolutionary reinforcement learning. In International conference on machine learning. PMLR, 3341-3350."},{"key":"e_1_3_2_1_21_1","volume-title":"Advances in Neural Information Processing Systems","volume":"31","author":"Khadka Shauharda","year":"2018","unstructured":"Shauharda Khadka and Kagan Tumer. 2018. Evolution-guided policy gradient in reinforcement learning. Advances in Neural Information Processing Systems, Vol. 31 (2018)."},{"key":"e_1_3_2_1_22_1","volume-title":"Pgps: Coupling policy gradient with population-based search.","author":"Kim Namyong","year":"2020","unstructured":"Namyong Kim, Hyunsuk Baek, and Hayong Shin. 2020. Pgps: Coupling policy gradient with population-based search. (2020)."},{"key":"e_1_3_2_1_23_1","volume-title":"Actor-critic algorithms. Advances in neural information processing systems","author":"Konda Vijay","year":"1999","unstructured":"Vijay Konda and John Tsitsiklis. 1999. Actor-critic algorithms. Advances in neural information processing systems, Vol. 12 (1999)."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2022.03.003"},{"key":"e_1_3_2_1_25_1","volume-title":"Kashu Yamazaki, Khoa Luu, and Marios Savvides.","author":"Le Ngan","year":"2022","unstructured":"Ngan Le, Vidhiwar Singh Rathour, Kashu Yamazaki, Khoa Luu, and Marios Savvides. 2022. Deep reinforcement learning in computer vision: a comprehensive survey. Artificial Intelligence Review (2022), 1-87."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/2001576.2001606"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1162\/isal_a_00338"},{"key":"e_1_3_2_1_28_1","volume-title":"Bridging evolutionary algorithms and reinforcement learning: A comprehensive survey on hybrid algorithms","author":"Li Pengyi","year":"2024","unstructured":"Pengyi Li, Jianye Hao, Hongyao Tang, Xian Fu, Yan Zhen, and Ke Tang. 2024a. Bridging evolutionary algorithms and reinforcement learning: A comprehensive survey on hybrid algorithms. IEEE Transactions on Evolutionary Computation (2024)."},{"key":"e_1_3_2_1_29_1","volume-title":"Value-Evolutionary-Based Reinforcement Learning. In Forty-first International Conference on Machine Learning.","author":"Li Pengyi","year":"2024","unstructured":"Pengyi Li, Jianye HAO, Hongyao Tang, YAN ZHENG, and Fazl Barez. 2024b. Value-Evolutionary-Based Reinforcement Learning. In Forty-first International Conference on Machine Learning."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11227-024-06377-2"},{"key":"e_1_3_2_1_31_1","volume-title":"Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971","author":"Lillicrap TP","year":"2015","unstructured":"TP Lillicrap. 2015. Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971 (2015)."},{"key":"e_1_3_2_1_32_1","volume-title":"Evolutionary Reinforcement Learning: A Systematic Review and Future Directions. arXiv preprint arXiv:2402.13296","author":"Lin Yuanguo","year":"2024","unstructured":"Yuanguo Lin, Fan Lin, Guorong Cai, Hong Chen, Lixin Zou, and Pengcheng Wu. 2024. Evolutionary Reinforcement Learning: A Systematic Review and Future Directions. arXiv preprint arXiv:2402.13296 (2024)."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1155\/2021\/5300189"},{"key":"e_1_3_2_1_34_1","volume-title":"Deep reinforcement learning versus evolution strategies: A comparative survey","author":"Majid Amjad Yousef","year":"2023","unstructured":"Amjad Yousef Majid, Serge Saaybi, Vincent Francois-Lavet, R Venkatesha Prasad, and Chris Verhoeven. 2023. Deep reinforcement learning versus evolution strategies: A comparative survey. IEEE Transactions on Neural Networks and Learning Systems (2023)."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3512290.3528845"},{"key":"e_1_3_2_1_36_1","volume-title":"International Conference on Learning Representations.","author":"Sigaud Pourchot","year":"2019","unstructured":"Pourchot and Sigaud. 2019. CEM-RL: Combining evolutionary and gradient-based methods for policy search. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_37_1","volume-title":"Markov decision processes. Handbooks in operations research and management science","author":"Puterman Martin L","year":"1990","unstructured":"Martin L Puterman. 1990. Markov decision processes. Handbooks in operations research and management science, Vol. 2 (1990), 331-434."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01224"},{"key":"e_1_3_2_1_39_1","volume-title":"Sr-cis: Self-reflective incremental system with decoupled memory and reasoning. arXiv preprint arXiv:2408.01970","author":"Qi Biqing","year":"2024","unstructured":"Biqing Qi, Junqi Gao, Xinquan Chen, Dong Li, Weinan Zhang, and Bowen Zhou. 2024b. Sr-cis: Self-reflective incremental system with decoupled memory and reasoning. arXiv preprint arXiv:2408.01970 (2024)."},{"key":"e_1_3_2_1_40_1","volume-title":"Large language models are zero shot hypothesis proposers. arXiv preprint arXiv:2311.05965","author":"Qi Biqing","year":"2023","unstructured":"Biqing Qi, Kaiyan Zhang, Haoxiang Li, Kai Tian, Sihang Zeng, Zhang-Ren Chen, and Bowen Zhou. 2023. Large language models are zero shot hypothesis proposers. arXiv preprint arXiv:2311.05965 (2023)."},{"key":"e_1_3_2_1_41_1","volume-title":"Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347","author":"Schulman John","year":"2017","unstructured":"John Schulman, Filip Wolski, Prafulla Dhariwal, Alec Radford, and Oleg Klimov. 2017. Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2023.120495"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3569096"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.swevo.2023.101236"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.swevo.2024.101517"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3467477"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386109"},{"key":"e_1_3_2_1_48_1","volume-title":"General subpopulation framework and taming the conflict inside populations. Evolutionary computation","author":"Vargas Danilo Vasconcellos","year":"2015","unstructured":"Danilo Vasconcellos Vargas, Junichi Murata, Hirotaka Takano, and Alexandre Cl\u00e1udio Botazzo Delbem. 2015. General subpopulation framework and taming the conflict inside populations. Evolutionary computation, Vol. 23, 1 (2015), 1-36."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICGTSPICC.2016.7955308"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.5555\/3692070.3694198"},{"key":"e_1_3_2_1_51_1","volume-title":"Simple statistical gradient-following algorithms for connectionist reinforcement learning. Machine learning","author":"Williams Ronald J","year":"1992","unstructured":"Ronald J Williams. 1992. Simple statistical gradient-following algorithms for connectionist reinforcement learning. Machine learning, Vol. 8 (1992), 229-256."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1007\/s40747-023-01243-9"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.asoc.2021.107494"},{"volume-title":"Evolutionary learning: Advances in theories and algorithms","author":"Zhou Zhi-Hua","key":"e_1_3_2_1_54_1","unstructured":"Zhi-Hua Zhou, Yang Yu, and Chao Qian. 2019. Evolutionary learning: Advances in theories and algorithms. Springer."},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i18.30079"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2023.126628"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i15.29665"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755206","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:53:58Z","timestamp":1765310038000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755206"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":57,"alternative-id":["10.1145\/3746027.3755206","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755206","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}