{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,28]],"date-time":"2026-07-28T10:51:11Z","timestamp":1785235871378,"version":"3.55.0"},"reference-count":61,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Swarm and Evolutionary Computation"],"published-print":{"date-parts":[[2026,3]]},"DOI":"10.1016\/j.swevo.2026.102330","type":"journal-article","created":{"date-parts":[[2026,2,17]],"date-time":"2026-02-17T13:24:44Z","timestamp":1771334684000},"page":"102330","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":1,"special_numbering":"C","title":["CMA-MAPPO: Integrating Covariance Matrix Adaptation Evolution Strategy with Multi-Agent Proximal Policy Optimization for enhanced exploration in sparse-reward environments"],"prefix":"10.1016","volume":"102","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-9045-6635","authenticated-orcid":false,"given":"A.H.","family":"Khatami","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"6","key":"10.1016\/j.swevo.2026.102330_b1","doi-asserted-by":"crossref","first-page":"5023","DOI":"10.1007\/s10462-022-10299-x","article-title":"Deep multiagent reinforcement learning: challenges and directions","volume":"56","author":"Wong","year":"2023","journal-title":"Artif. Intell. Rev."},{"key":"10.1016\/j.swevo.2026.102330_b2","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1016\/j.inffus.2022.03.003","article-title":"Exploration in deep reinforcement learning: A survey","volume":"85","author":"Ladosz","year":"2022","journal-title":"Inf. Fusion"},{"issue":"1","key":"10.1016\/j.swevo.2026.102330_b3","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1162\/106365603321828970","article-title":"Reducing the time complexity of the derandomized evolution strategy with covariance matrix adaptation (cma-es)","volume":"11","author":"Hansen","year":"2003","journal-title":"Evol. Comput."},{"key":"10.1016\/j.swevo.2026.102330_b4","first-page":"24611","article-title":"The surprising effectiveness of ppo in cooperative multi-agent games","volume":"35","author":"Yu","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.swevo.2026.102330_b5","doi-asserted-by":"crossref","DOI":"10.1016\/j.swevo.2023.101351","article-title":"Reinforcement learning-based hybrid differential evolution for global optimization of interplanetary trajectory design","volume":"81","author":"Peng","year":"2023","journal-title":"Swarm Evol. Comput."},{"key":"10.1016\/j.swevo.2026.102330_b6","series-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017"},{"issue":"2","key":"10.1016\/j.swevo.2026.102330_b7","first-page":"73","article-title":"A survey on multi-agent reinforcement learning and its application","volume":"3","author":"Ning","year":"2024","journal-title":"J. Autom. Intell."},{"issue":"181","key":"10.1016\/j.swevo.2026.102330_b8","first-page":"1","article-title":"Curriculum learning for reinforcement learning domains: A framework and survey","volume":"21","author":"Narvekar","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.swevo.2026.102330_b9","series-title":"Generating large-scale dynamic optimization problem instances using the generalized moving peaks benchmark","author":"Omidvar","year":"2021"},{"key":"10.1016\/j.swevo.2026.102330_b10","series-title":"Exploration by random network distillation","author":"Burda","year":"2018"},{"key":"10.1016\/j.swevo.2026.102330_b11","series-title":"Conference on Robot Learning","first-page":"835","article-title":"Teacher algorithms for curriculum learning of deep rl in continuously parameterized environments","author":"Portelas","year":"2020"},{"key":"10.1016\/j.swevo.2026.102330_b12","doi-asserted-by":"crossref","first-page":"241","DOI":"10.1613\/jair.613","article-title":"Evolutionary algorithms for reinforcement learning","volume":"11","author":"Moriarty","year":"1999","journal-title":"J. Artificial Intelligence Res."},{"key":"10.1016\/j.swevo.2026.102330_b13","volume":"vol. 1","author":"Oliehoek","year":"2016"},{"key":"10.1016\/j.swevo.2026.102330_b14","series-title":"High-dimensional continuous control using generalized advantage estimation","author":"Schulman","year":"2015"},{"key":"10.1016\/j.swevo.2026.102330_b15","series-title":"Revisiting some common practices in cooperative multi-agent reinforcement learning","author":"Fu","year":"2022"},{"key":"10.1016\/j.swevo.2026.102330_b16","doi-asserted-by":"crossref","DOI":"10.3389\/fnins.2023.1201370","article-title":"A semi-independent policies training method with shared representation for heterogeneous multi-agents reinforcement learning","volume":"17","author":"Zhao","year":"2023","journal-title":"Front. Neurosci."},{"key":"10.1016\/j.swevo.2026.102330_b17","doi-asserted-by":"crossref","first-page":"0025","DOI":"10.34133\/icomputing.0025","article-title":"Evolutionary reinforcement learning: A survey","volume":"2","author":"Bai","year":"2023","journal-title":"Intell. Comput."},{"issue":"2","key":"10.1016\/j.swevo.2026.102330_b18","doi-asserted-by":"crossref","first-page":"159","DOI":"10.1162\/106365601750190398","article-title":"Completely derandomized self-adaptation in evolution strategies","volume":"9","author":"Hansen","year":"2001","journal-title":"Evol. Comput."},{"key":"10.1016\/j.swevo.2026.102330_b19","article-title":"Hindsight experience replay","volume":"vol. 30","author":"Andrychowicz","year":"2017"},{"key":"10.1016\/j.swevo.2026.102330_b20","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.123289","article-title":"Covariance matrix adaptation evolution strategy based on correlated evolution paths with application to reinforcement learning","volume":"246","author":"Ajani","year":"2024","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.swevo.2026.102330_b21","doi-asserted-by":"crossref","unstructured":"M. Tan, Multi-agent reinforcement learning: Independent vs. cooperative agents, in: Proceedings of the Tenth International Conference on Machine Learning, 1993, pp. 330\u2013337.","DOI":"10.1016\/B978-1-55860-307-3.50049-6"},{"key":"10.1016\/j.swevo.2026.102330_b22","series-title":"Value-decomposition networks for cooperative multi-agent learning","author":"Sunehag","year":"2017"},{"issue":"178","key":"10.1016\/j.swevo.2026.102330_b23","first-page":"1","article-title":"Monotonic value function factorisation for deep multi-agent reinforcement learning","volume":"21","author":"Rashid","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.swevo.2026.102330_b24","article-title":"Multi-agent actor-critic for mixed cooperative-competitive environments","volume":"vol. 30","author":"Lowe","year":"2017"},{"key":"10.1016\/j.swevo.2026.102330_b25","series-title":"Is centralized training with decentralized execution framework centralized enough for marl?","author":"Zhou","year":"2023"},{"key":"10.1016\/j.swevo.2026.102330_b26","series-title":"ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"6035","article-title":"Adaptive parameter sharing for multi-agent reinforcement learning","author":"Li","year":"2024"},{"key":"10.1016\/j.swevo.2026.102330_b27","doi-asserted-by":"crossref","unstructured":"J. Wang, Q. Hao, W. Huang, X. Fan, Z. Tang, B. Wang, J. Hao, Y. Li, Dyps: Dynamic parameter sharing in multi-agent reinforcement learning for spatio-temporal resource allocation, in: Proceedings of the 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining, 2024, pp. 3128\u20133139.","DOI":"10.1145\/3637528.3672052"},{"key":"10.1016\/j.swevo.2026.102330_b28","unstructured":"Y. Yu, Q. Yin, J. Zhang, P. Xu, K. Huang, Admn: Agent-driven modular network for dynamic parameter sharing in cooperative multi-agent reinforcement learning, in: Proceedings of the Thirty-Third International Joint Conference on Artificial Intelligence, IJCAI-24, 2024, pp. 302\u2013310."},{"key":"10.1016\/j.swevo.2026.102330_b29","series-title":"Proceedings of the 34th International Conference on Machine Learning","first-page":"2778","article-title":"Curiosity-driven exploration by self-supervised prediction","volume":"vol. 70","author":"Pathak","year":"2017"},{"key":"10.1016\/j.swevo.2026.102330_b30","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2025.110866","article-title":"Hierarchical reinforcement learning with curriculum demonstrations and goal-guided policies for sequential robotic manipulation","volume":"153","author":"Sun","year":"2025","journal-title":"Eng. Appl. Artif. Intell."},{"issue":"3","key":"10.1016\/j.swevo.2026.102330_b31","doi-asserted-by":"crossref","first-page":"1107","DOI":"10.1007\/s10845-023-02094-4","article-title":"A two-stage rnn-based deep reinforcement learning approach for solving the parallel machine scheduling problem with due dates and family setups","volume":"35","author":"Li","year":"2024","journal-title":"J. Intell. Manuf."},{"issue":"7","key":"10.1016\/j.swevo.2026.102330_b32","doi-asserted-by":"crossref","first-page":"4735","DOI":"10.1007\/s10845-024-02470-8","article-title":"A transformer-based deep reinforcement learning approach for dynamic parallel machine scheduling problem with family setups","volume":"36","author":"Li","year":"2025","journal-title":"Journal of Intelligent Manufacturing"},{"key":"10.1016\/j.swevo.2026.102330_b33","doi-asserted-by":"crossref","DOI":"10.1016\/j.ijnaoe.2024.100629","article-title":"Simulation-based deep reinforcement learning for multi-objective identical parallel machine scheduling problem","volume":"16","author":"Nam","year":"2024","journal-title":"Int. J. Nav. Archit. Ocean. Eng."},{"key":"10.1016\/j.swevo.2026.102330_b34","doi-asserted-by":"crossref","DOI":"10.1016\/j.swevo.2024.101808","article-title":"A revised deep reinforcement learning algorithm for parallel machine scheduling problem under multi-scenario due date constraints","volume":"92","author":"Zhang","year":"2025","journal-title":"Swarm Evol. Comput."},{"key":"10.1016\/j.swevo.2026.102330_b35","article-title":"Proximal policy optimization with population-based variable neighborhood search algorithm for coordinating photo-etching and acid-etching processes in sustainable storage chip manufacturing","volume":"42","author":"Zhang","year":"2024","journal-title":"J. Ind. Inf. Integr."},{"key":"10.1016\/j.swevo.2026.102330_b36","doi-asserted-by":"crossref","DOI":"10.1016\/j.swevo.2025.101944","article-title":"An evolution strategies-based reinforcement learning algorithm for multi-objective dynamic parallel machine scheduling problems","volume":"95","author":"Chen","year":"2025","journal-title":"Swarm Evol. Comput."},{"key":"10.1016\/j.swevo.2026.102330_b37","series-title":"Optimizing uav aerial base station flights using drl-based proximal policy optimization","author":"Ib\u00e1nez","year":"2025"},{"key":"10.1016\/j.swevo.2026.102330_b38","series-title":"Drl-based injection molding process parameter optimization for adaptive and profitable production","author":"Kim","year":"2025"},{"key":"10.1016\/j.swevo.2026.102330_b39","series-title":"2025 8th International Conference on Circuits, Systems and Simulation","first-page":"90","article-title":"Drl-based approach for ris-aided noma system in short packet communications","author":"Khalid","year":"2025"},{"key":"10.1016\/j.swevo.2026.102330_b40","doi-asserted-by":"crossref","first-page":"117","DOI":"10.1016\/j.cor.2018.07.002","article-title":"Exact algorithms for the order picking problem","volume":"100","author":"Pansart","year":"2018","journal-title":"Comput. Oper. Res."},{"key":"10.1016\/j.swevo.2026.102330_b41","series-title":"2020 IEEE 30th International Workshop on Machine Learning for Signal Processing","first-page":"1","article-title":"Ppo-cma: Proximal policy optimization with covariance matrix adaptation","author":"H\u00e4m\u00e4l\u00e4inen","year":"2020"},{"key":"10.1016\/j.swevo.2026.102330_b42","series-title":"2023 35th Chinese Control and Decision Conference","first-page":"4847","article-title":"Covariance matrix adaptation for multi-agent proximal policy optimization","author":"Shen","year":"2023"},{"issue":"3","key":"10.1016\/j.swevo.2026.102330_b43","doi-asserted-by":"crossref","first-page":"445","DOI":"10.1007\/s12293-024-00419-1","article-title":"Proximal evolutionary strategy: improving deep reinforcement learning through evolutionary policy optimization","volume":"16","author":"Peng","year":"2024","journal-title":"Memetic Comput."},{"issue":"1","key":"10.1016\/j.swevo.2026.102330_b44","doi-asserted-by":"crossref","first-page":"449","DOI":"10.1007\/s00521-022-07779-0","article-title":"Policy-based optimization: single-step policy gradient method seen as an evolution strategy","volume":"35","author":"Viquerat","year":"2023","journal-title":"Neural Comput. Appl."},{"key":"10.1016\/j.swevo.2026.102330_b45","doi-asserted-by":"crossref","unstructured":"T. Xu, H.C. Chen, J. He, Accelerate evolution strategy by proximal policy optimization, in: Proceedings of the Genetic and Evolutionary Computation Conference, 2024, pp. 1064\u20131072.","DOI":"10.1145\/3638529.3654090"},{"issue":"20","key":"10.1016\/j.swevo.2026.102330_b46","doi-asserted-by":"crossref","first-page":"12141","DOI":"10.1007\/s00500-024-09913-7","article-title":"Monitoring uav status and detecting insulator faults in transmission lines with a new classifier based on aggregation votes between neural networks by interval type-2 tsk fuzzy system","volume":"28","author":"Amiri","year":"2024","journal-title":"Soft Comput."},{"issue":"5","key":"10.1016\/j.swevo.2026.102330_b47","doi-asserted-by":"crossref","first-page":"1707","DOI":"10.1109\/TEVC.2024.3443913","article-title":"Bridging evolutionary algorithms and reinforcement learning: A comprehensive survey on hybrid algorithms","volume":"29","author":"Li","year":"2025","journal-title":"IEEE Trans. Evol. Comput."},{"key":"10.1016\/j.swevo.2026.102330_b48","series-title":"Deep neuroevolution: Genetic algorithms are a competitive alternative for training deep neural networks for reinforcement learning","author":"Such","year":"2017"},{"issue":"5","key":"10.1016\/j.swevo.2026.102330_b49","doi-asserted-by":"crossref","first-page":"830","DOI":"10.1109\/TEVC.2021.3061466","article-title":"As-nas: Adaptive scalable neural architecture search with reinforced evolutionary algorithm for deep learning","volume":"25","author":"Zhang","year":"2021","journal-title":"IEEE Trans. Evol. Comput."},{"key":"10.1016\/j.swevo.2026.102330_b50","series-title":"International Conference on Machine Learning","first-page":"4264","article-title":"Guided evolutionary strategies: Augmenting random search with surrogate gradients","author":"Maheswaranathan","year":"2019"},{"issue":"3","key":"10.1016\/j.swevo.2026.102330_b51","first-page":"171","article-title":"Evolutionary reinforcement learning of artificial neural networks","volume":"4","author":"Siebel","year":"2007","journal-title":"Int. J. Hybrid Intell. Syst."},{"issue":"3","key":"10.1016\/j.swevo.2026.102330_b52","doi-asserted-by":"crossref","first-page":"333","DOI":"10.3233\/IDA-2010-0424","article-title":"Learning hybridization strategies in evolutionary algorithms","volume":"14","author":"LaTorre","year":"2010","journal-title":"Intell. Data Anal."},{"issue":"1","key":"10.1016\/j.swevo.2026.102330_b53","doi-asserted-by":"crossref","first-page":"5032","DOI":"10.1038\/s41598-024-54910-3","article-title":"Hippopotamus optimization algorithm: a novel nature-inspired optimization algorithm","volume":"14","author":"Amiri","year":"2024","journal-title":"Sci. Rep."},{"key":"10.1016\/j.swevo.2026.102330_b54","article-title":"Q2ho-mftv: A binary hippopotamus optimization algorithm for feature selection with a brief review of binary optimization","author":"Hashjin","year":"2025","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.swevo.2026.102330_b55","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.125497","article-title":"An innovative data-driven ai approach for detecting and isolating faults in gas turbines at power plants","volume":"263","author":"Amiri","year":"2025","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.swevo.2026.102330_b56","series-title":"International Conference on Machine Learning","first-page":"6651","article-title":"Evolutionary reinforcement learning for sample-efficient multiagent coordination","author":"Majumdar","year":"2020"},{"key":"10.1016\/j.swevo.2026.102330_b57","series-title":"Proceedings of the 7th Annual Workshop on Genetic and Evolutionary Computation, GECCO \u201905","first-page":"39","article-title":"Learning, anticipation and time-deception in evolutionary online dynamic optimization","author":"Bosman","year":"2005"},{"key":"10.1016\/j.swevo.2026.102330_b58","doi-asserted-by":"crossref","DOI":"10.1016\/j.swevo.2022.101125","article-title":"A new moving peaks benchmark with attractors for dynamic evolutionary algorithms","volume":"74","author":"Fox","year":"2022","journal-title":"Swarm Evol. Comput."},{"issue":"3","key":"10.1016\/j.swevo.2026.102330_b59","article-title":"Olympus: a benchmarking framework for noisy optimization and experiment planning","volume":"2","author":"H\u00e4se","year":"2021","journal-title":"Mach. Learn.: Sci. Technol."},{"issue":"4","key":"10.1016\/j.swevo.2026.102330_b60","doi-asserted-by":"crossref","first-page":"341","DOI":"10.1023\/A:1008202821328","article-title":"Differential evolution - a simple and efficient heuristic for global optimization over continuous spaces","volume":"11","author":"Storn","year":"1997","journal-title":"J. Global Optim."},{"issue":"7","key":"10.1016\/j.swevo.2026.102330_b61","doi-asserted-by":"crossref","first-page":"10197","DOI":"10.1007\/s10586-024-04475-7","article-title":"Novel hybrid classifier based on fuzzy type-iii decision maker and ensemble deep learning model and improved chaos game optimization","volume":"27","author":"Mehrabi Hashjin","year":"2024","journal-title":"Clust. Comput."}],"container-title":["Swarm and Evolutionary Computation"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S2210650226000507?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S2210650226000507?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T16:19:33Z","timestamp":1773159573000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S2210650226000507"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3]]},"references-count":61,"alternative-id":["S2210650226000507"],"URL":"https:\/\/doi.org\/10.1016\/j.swevo.2026.102330","relation":{},"ISSN":["2210-6502"],"issn-type":[{"value":"2210-6502","type":"print"}],"subject":[],"published":{"date-parts":[[2026,3]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"CMA-MAPPO: Integrating Covariance Matrix Adaptation Evolution Strategy with Multi-Agent Proximal Policy Optimization for enhanced exploration in sparse-reward environments","name":"articletitle","label":"Article Title"},{"value":"Swarm and Evolutionary Computation","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.swevo.2026.102330","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"102330"}}