{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T04:08:42Z","timestamp":1785470922469,"version":"3.56.0"},"reference-count":199,"publisher":"Springer Science and Business Media LLC","issue":"11","license":[{"start":{"date-parts":[[2023,3,25]],"date-time":"2023-03-25T00:00:00Z","timestamp":1679702400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,3,25]],"date-time":"2023-03-25T00:00:00Z","timestamp":1679702400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Artif Intell Rev"],"published-print":{"date-parts":[[2023,11]]},"DOI":"10.1007\/s10462-023-10468-6","type":"journal-article","created":{"date-parts":[[2023,3,25]],"date-time":"2023-03-25T11:03:08Z","timestamp":1679742188000},"page":"12885-12947","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":50,"title":["Reinforcement learning for predictive maintenance: a systematic technical review"],"prefix":"10.1007","volume":"56","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2341-8787","authenticated-orcid":false,"given":"Rajesh","family":"Siraskar","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Satish","family":"Kumar","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shruti","family":"Patil","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Arunkumar","family":"Bongale","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ketan","family":"Kotecha","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,3,25]]},"reference":[{"key":"10468_CR1","unstructured":"Abernethy RB (2018) Dr. E. H. Wallodi Weibull. http:\/\/km.fgg.uni-lj.si\/PREDMETI\/sei\/Ljudje\/weibull.htm"},{"key":"10468_CR2","unstructured":"Abudali M, Siegel D (2021) A pressing case for predictive analytics at Maclean\u2013Fogg. https:\/\/www.plantengineering.com\/articles\/a-pressing-case-for-predictive-analytics-at-maclean-fogg\/"},{"key":"10468_CR3","unstructured":"Achiam J (2018a) Deep deterministic policy gradient\u2014the q-learning side of DDPG. https:\/\/spinningup.openai.com\/en\/latest\/algorithms\/ddpg.html#the-q-learning-side-of-ddpg"},{"key":"10468_CR4","unstructured":"Achiam J (2018b) Part 1: key concepts in RL\u2014spinning up documentation. OpenAI. https:\/\/spinningup.openai.com\/en\/latest\/spinningup\/rl_intro.html#key-concepts-and-terminology"},{"key":"10468_CR5","doi-asserted-by":"publisher","unstructured":"Adams S, Meekins R, Beling P et al (2019) Hierarchical fault classification for resource constrained systems. Mech Syst Signal Process. https:\/\/doi.org\/10.1016\/j.ymssp.2019.106266","DOI":"10.1016\/j.ymssp.2019.106266"},{"issue":"4","key":"10468_CR6","doi-asserted-by":"publisher","first-page":"182","DOI":"10.1049\/IET-CIM.2020.0022","volume":"2","author":"A Adsule","year":"2020","unstructured":"Adsule A, Kulkarni M, Tewari A (2020) Reinforcement learning for optimal policy learning in condition-based maintenance. IET Collabor Intell Manuf 2(4):182\u2013188. https:\/\/doi.org\/10.1049\/IET-CIM.2020.0022","journal-title":"IET Collaborative Intelligent Manufacturing"},{"key":"10468_CR7","doi-asserted-by":"publisher","unstructured":"Afshari H, Al-Ani D, Habibi S (2014) Fault prognosis of roller bearings using the adaptive auto-step reinforcement learning technique. In: ASME 2014 dynamic systems and control conference (DSCC 20140), p 1. https:\/\/doi.org\/10.1115\/dscc2014-5928","DOI":"10.1115\/dscc2014-5928"},{"issue":"24","key":"10468_CR8","doi-asserted-by":"publisher","first-page":"233","DOI":"10.1016\/j.ifacol.2018.09.583","volume":"51","author":"I Ahmed","year":"2018","unstructured":"Ahmed I, Khorasgani H, Biswas G (2018) Comparison of model predictive and reinforcement learning methods for fault tolerant control. IFAC-Papers OnLine 51(24):233\u2013240","journal-title":"IFAC-PapersOnLine"},{"issue":"7","key":"10468_CR9","doi-asserted-by":"publisher","first-page":"1089","DOI":"10.1016\/j.engappai.2009.01.014","volume":"22","author":"N Aissani","year":"2009","unstructured":"Aissani N, Beldjilali B, Trentesaux D (2009) Dynamic scheduling of maintenance tasks in the petroleum industry: a reinforcement approach. Eng Appl Artif Intell 22(7):1089\u20131103. https:\/\/doi.org\/10.1016\/j.engappai.2009.01.014","journal-title":"Engineering Applications of Artificial Intelligence"},{"key":"10468_CR10","doi-asserted-by":"crossref","unstructured":"Alimi M, Rhif A, Rebai A et\u00a0al (2021) Optimal adaptive backstepping control for chaos synchronization of nonlinear dynamical systems. Backstepping Control of Nonlinear Dynamical Systems pp 291\u2013345","DOI":"10.1016\/B978-0-12-817582-8.00020-9"},{"key":"10468_CR11","doi-asserted-by":"publisher","unstructured":"Andriotis C, Papakonstantinou K (2019) Managing engineering systems with large state and action spaces through deep reinforcement learning. Reliab Eng Syst Saf 191:106483. https:\/\/doi.org\/10.1016\/j.ress.2019.04.036","DOI":"10.1016\/j.ress.2019.04.036"},{"issue":"107","key":"10468_CR12","first-page":"551","volume":"212","author":"C Andriotis","year":"2021","unstructured":"Andriotis C, Papakonstantinou K (2021) Deep reinforcement learning driven inspection and maintenance planning under incomplete information and constraints. Reliab Eng Syst Saf 212(107):551","journal-title":"Reliability Engineering & System Safety"},{"key":"10468_CR13","doi-asserted-by":"publisher","unstructured":"Bala R, Govinda R, Murthy CS (2018) Reliability analysis and failure rate evaluation of load haul dump machines using weibull distribution analysis. Math Model 5(2):116\u2013122. https:\/\/doi.org\/10.18280\/mmep.050209","DOI":"10.18280\/mmep.050209"},{"issue":"1","key":"10468_CR14","doi-asserted-by":"publisher","first-page":"147","DOI":"10.1007\/s10845-016-1237-7","volume":"30","author":"S Barde","year":"2019","unstructured":"Barde S, Yacout S, Shin H (2019) Optimal preventive maintenance policy based on reinforcement learning of a fleet of military trucks. J Intell Manuf 30(1):147\u2013161. https:\/\/doi.org\/10.1007\/s10845-016-1237-7","journal-title":"Journal of Intelligent Manufacturing"},{"issue":"111","key":"10468_CR15","doi-asserted-by":"publisher","first-page":"459","DOI":"10.1016\/j.rser.2021.111459","volume":"150","author":"S Barja-Martinez","year":"2021","unstructured":"Barja-Martinez S, Arag\u00fc\u00e9s-Pe\u00f1alba M, Munn\u00e9-Collado \u00cd et al (2021) Artificial intelligence techniques for enabling big data services in distribution networks: a review. Renew Sustain Energy Rev 150(111):459. https:\/\/doi.org\/10.1016\/j.rser.2021.111459","journal-title":"Renewable and Sustainable Energy Reviews"},{"key":"10468_CR16","doi-asserted-by":"crossref","unstructured":"Baykal-G\u00fcrsoy M (2010) Semi-markov decision processes. In: Wiley encyclopedia of operations research and management science. Wiley, Hoboken","DOI":"10.1002\/9780470400531.eorms0757"},{"key":"10468_CR17","doi-asserted-by":"publisher","unstructured":"Bellani L, Compare M, Baraldi P et al (2019) Towards developing a novel framework for practical PHM: a sequential decision problem solved by reinforcement learning and artificial neural networks. Int J Progn Health Manag 10(4). https:\/\/doi.org\/10.36001\/ijphm.2019.v10i4.2616","DOI":"10.36001\/ijphm.2019.v10i4.2616"},{"key":"10468_CR18","volume-title":"Maintenance, modeling and optimization","author":"M Ben-Daya","year":"2012","unstructured":"Ben-Daya M, Duffuaa SO, Raouf A (2012) Maintenance, modeling and optimization. Springer, Berlin"},{"key":"10468_CR19","unstructured":"Burke R, Mussomeli A, Laaper S et\u00a0al (2017) The smart factory. Deloitte Insights. https:\/\/www2.deloitte.com\/us\/en\/insights\/focus\/industry-4-0\/smart-factory-connected-manufacturing.html"},{"key":"10468_CR20","doi-asserted-by":"publisher","DOI":"10.1201\/9781439821091","volume-title":"Reinforcement learning and dynamic programming using function approximators","author":"L Busoniu","year":"2017","unstructured":"Busoniu L, Babuska R, De Schutter B et al (2017) Reinforcement learning and dynamic programming using function approximators. CRC Press, Boca Raton"},{"issue":"1","key":"10468_CR21","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1080\/03610927408827101","volume":"3","author":"T Cali\u0144ski","year":"1974","unstructured":"Cali\u0144ski T, Harabasz J (1974) A dendrite method for cluster analysis. Commun Stat Theory Methods 3(1):1\u201327","journal-title":"Communications in Statistics-theory and Methods"},{"key":"10468_CR22","unstructured":"Chen H, Li X (2011) Distributed active learning with application to battery health management. In: 14th International conference on information fusion 2011"},{"issue":"3","key":"10468_CR23","doi-asserted-by":"publisher","first-page":"2521","DOI":"10.1109\/TIE.2020.2972443","volume":"68","author":"Z Chen","year":"2020","unstructured":"Chen Z, Wu M, Zhao R et al (2020) Machine remaining useful life prediction via an attention-based deep learning approach. IEEE Trans Ind Electron 68(3):2521\u20132531","journal-title":"IEEE Transactions on Industrial Electronics"},{"issue":"5","key":"10468_CR24","doi-asserted-by":"publisher","first-page":"4393","DOI":"10.1109\/TIE.2020.2984976","volume":"68","author":"G Chen","year":"2021","unstructured":"Chen G, Liu M, Kong Z (2021) Temporal-logic-based semantic fault diagnosis with time-series data from industrial Internet of Things. IEEE Trans Ind Electron 68(5):4393\u20134403. https:\/\/doi.org\/10.1109\/TIE.2020.2984976","journal-title":"IEEE Transactions on Industrial Electronics"},{"issue":"1","key":"10468_CR25","doi-asserted-by":"publisher","first-page":"484","DOI":"10.1109\/TR.2020.3044596","volume":"71","author":"Y Chen","year":"2022","unstructured":"Chen Y, Liu Y, Xiahou T (2022) A deep reinforcement learning approach to dynamic loading strategy of repairable multistate systems. IEEE Trans Reliab 71(1):484\u2013499. https:\/\/doi.org\/10.1109\/TR.2020.3044596","journal-title":"IEEE Transactions on Reliability"},{"key":"10468_CR26","doi-asserted-by":"publisher","unstructured":"Cheng M, Frangopol D (2021) A decision-making framework for load rating planning of aging bridges using deep reinforcement learning. J Comput Civ Eng. https:\/\/doi.org\/10.1061\/(ASCE)CP.1943-5487.0000991","DOI":"10.1061\/(ASCE)CP.1943-5487.0000991"},{"key":"10468_CR27","doi-asserted-by":"publisher","unstructured":"Cheng Y, Peng J, Gu X et\u00a0al (2018) RLCP: a reinforcement learning method for health stage division using change points. In: 2018 IEEE international conference on prognostics and health management (ICPHM 2018). https:\/\/doi.org\/10.1109\/ICPHM.2018.8448499","DOI":"10.1109\/ICPHM.2018.8448499"},{"key":"10468_CR28","unstructured":"Coleman C, Damodaran S, Deuel E (2017) Predictive maintenance and the smart factory. https:\/\/www2.deloitte.com\/content\/dam\/Deloitte\/us\/Documents\/process-and-operations\/us-cons-predictive-maintenance.pdf"},{"issue":"1","key":"10468_CR29","doi-asserted-by":"publisher","first-page":"52","DOI":"10.1177\/1748006X19869750","volume":"234","author":"M Compare","year":"2020","unstructured":"Compare M, Bellani L, Cobelli E et al (2020) A reinforcement learning approach to optimal part flow management for gas turbine maintenance. Proc Inst Mech Eng Part O J Risk Reliab 234(1):52\u201362. https:\/\/doi.org\/10.1177\/1748006X19869750","journal-title":"Proceedings of the Institution of Mechanical Engineers, Part O: Journal of Risk and Reliability"},{"key":"10468_CR30","doi-asserted-by":"publisher","unstructured":"Correa JCAJ, Guzman AAL (2020) Guidelines for the implementation of a predictive maintenance program. Mech Vib Condit Monit. https:\/\/doi.org\/10.1016\/B978-0-12-819796-7.00007-X","DOI":"10.1016\/B978-0-12-819796-7.00007-X"},{"key":"10468_CR31","doi-asserted-by":"publisher","unstructured":"Correa-Jullian C, Droguett EL, Cardemil JM (2020) Operation scheduling in a solar thermal system: a reinforcement learning-based framework. Appl Energy. https:\/\/doi.org\/10.1016\/j.apenergy.2020.114943","DOI":"10.1016\/j.apenergy.2020.114943"},{"issue":"12","key":"10468_CR32","doi-asserted-by":"publisher","first-page":"3416","DOI":"10.13196\/j.cims.2021.12.004","volume":"27","author":"P Cui","year":"2021","unstructured":"Cui P, Wang J, Zhang W et al (2021) Predictive maintenance decision-making for serial production lines based on deep reinforcement learning. Comput Integrated Manuf Syst (CIMS) 27(12):3416\u20133428. https:\/\/doi.org\/10.13196\/j.cims.2021.12.004","journal-title":"Jisuanji Jicheng Zhizao Xitong\/Computer Integrated Manufacturing Systems, CIMS"},{"issue":"22","key":"10468_CR33","doi-asserted-by":"publisher","first-page":"6848","DOI":"10.1080\/00207543.2021.1962558","volume":"60","author":"PH Cui","year":"2022","unstructured":"Cui PH, Wang JQ, Li Y (2022) Data-driven modelling, analysis and improvement of multistage production systems with predictive maintenance and product quality. Int J Prod Res 60(22):6848\u20136865","journal-title":"International Journal of Production Research"},{"key":"10468_CR34","unstructured":"Dahlqvist F, Patel M, Rajko A et\u00a0al (2019) Growing opportunities in the Internet Of Things. https:\/\/www.mckinsey.com\/industries\/private-equity-and-principal-investors\/our-insights\/growing-opportunities-in-the-internet-of-things"},{"issue":"15","key":"10468_CR35","doi-asserted-by":"publisher","first-page":"8307","DOI":"10.1109\/JSEN.2020.2970747","volume":"20","author":"W Dai","year":"2020","unstructured":"Dai W, Mo Z, Luo C et al (2020) Fault diagnosis of rotating machinery based on deep reinforcement learning and reciprocal of smoothness index. IEEE Sensors J 20(15):8307\u20138315. https:\/\/doi.org\/10.1109\/JSEN.2020.2970747","journal-title":"IEEE Sensors Journal"},{"key":"10468_CR36","doi-asserted-by":"publisher","unstructured":"Dai Z, Jiang M, Li X et al (2021) Reinforcement lion swarm optimization algorithm for tool wear prediction. In: 2021 Global reliability and prognostics and health management (PHM)\u2014Nanjing 2021. https:\/\/doi.org\/10.1109\/PHM-Nanjing52125.2021.9613134","DOI":"10.1109\/PHM-Nanjing52125.2021.9613134"},{"key":"10468_CR37","doi-asserted-by":"publisher","unstructured":"Dangut M, Jennions I, King S et al (2022) Application of deep reinforcement learning for extremely rare failure prediction in aircraft maintenance. Mech Syst Signal Process. https:\/\/doi.org\/10.1016\/j.ymssp.2022.108873","DOI":"10.1016\/j.ymssp.2022.108873"},{"issue":"4","key":"10468_CR38","doi-asserted-by":"publisher","first-page":"560","DOI":"10.1287\/mnsc.45.4.560","volume":"45","author":"T Das","year":"1999","unstructured":"Das T, Gosavi A, Mahadevan S et al (1999) Solving semi-Markov decision problems using average reward reinforcement learning. Manag Sci 45(4):560\u2013574. https:\/\/doi.org\/10.1287\/mnsc.45.4.560","journal-title":"Management Science"},{"key":"10468_CR39","doi-asserted-by":"crossref","unstructured":"Dau HA, Bagnall A, Kamgar K et\u00a0al (2019) The UCR time series archive. Mach Learn. arXiv:1810.07758","DOI":"10.1109\/JAS.2019.1911747"},{"key":"10468_CR40","doi-asserted-by":"crossref","unstructured":"Deloitte (2020) Industry 4.0. Deloitte Insights https:\/\/www2.deloitte.com\/us\/en\/insights\/focus\/industry-4-0.html","DOI":"10.1016\/j.focat.2020.03.003"},{"key":"10468_CR41","doi-asserted-by":"crossref","unstructured":"Ding F, He Z, Zi Y et\u00a0al (2008) Application of support vector machine for equipment reliability forecasting. In: 2008 6th IEEE international conference on industrial informatics, pp 526\u2013530","DOI":"10.1109\/INDIN.2008.4618157"},{"key":"10468_CR42","doi-asserted-by":"publisher","unstructured":"Ding Y, Ma L, Ma J et al (2019) Intelligent fault diagnosis for rotating machinery using deep Q-network based health state classification: a deep reinforcement learning approach. Adv Eng Inf. https:\/\/doi.org\/10.1016\/j.aei.2019.100977","DOI":"10.1016\/j.aei.2019.100977"},{"issue":"107","key":"10468_CR43","doi-asserted-by":"publisher","first-page":"760","DOI":"10.1016\/j.compchemeng.2022.107760","volume":"161","author":"O Dogru","year":"2022","unstructured":"Dogru O, Velswamy K, Ibrahim F et al (2022) Reinforcement learning approach to autonomous PID tuning. Comput Chem Eng 161(107):760. https:\/\/doi.org\/10.1016\/j.compchemeng.2022.107760","journal-title":"Computers and Chemical Engineering"},{"key":"10468_CR44","doi-asserted-by":"publisher","first-page":"343","DOI":"10.1016\/j.isatra.2020.09.004","volume":"108","author":"S Dong","year":"2021","unstructured":"Dong S, Wen G, Lei Z et al (2021a) Transfer learning for bearing performance degradation assessment based on deep hierarchical features. ISA Trans 108:343\u2013355. https:\/\/doi.org\/10.1016\/j.isatra.2020.09.004","journal-title":"ISA Transactions"},{"key":"10468_CR45","doi-asserted-by":"crossref","unstructured":"Dong W, Zhao T, Wu Y (2021b) Deep reinforcement learning based preventive maintenance for wind turbines. In: 2021 IEEE 5th conference on energy internet and energy system integration (EI2), pp 2860\u20132865","DOI":"10.1109\/EI252483.2021.9713457"},{"key":"10468_CR46","unstructured":"Duan Y, Chen X, Houthooft R et\u00a0al (2016) Benchmarking deep reinforcement learning for continuous control. arXiv:1604.06778"},{"key":"10468_CR47","doi-asserted-by":"publisher","unstructured":"Dulac-Arnold G, Levine N, Mankowitz DJ et al (2021) Challenges of real-world reinforcement learning: definitions, benchmarks and analysis. Mach Learn. https:\/\/doi.org\/10.1007\/s10994-021-05961-4","DOI":"10.1007\/s10994-021-05961-4"},{"key":"10468_CR48","doi-asserted-by":"crossref","unstructured":"Eke S, Aka-Ngnui T, Clerc G et\u00a0al (2017) Characterization of the operating periods of a power transformer by clustering the dissolved gas data. In: 2017 IEEE 11th International symposium on diagnostics for electrical machines, power electronics and drives (SDEMPED), pp 298\u2013303","DOI":"10.1109\/DEMPED.2017.8062371"},{"key":"10468_CR49","doi-asserted-by":"publisher","unstructured":"Eltotongy A, Awad M, Maged S et al (2021) Fault detection and classification of machinery bearing under variable operating conditions based on wavelet transform and CNN. 2021 International Mobile. Intelligent, and Ubiquitous Computing Conference, MIUCC 2021:117\u2013123. https:\/\/doi.org\/10.1109\/MIUCC52538.2021.9447673","DOI":"10.1109\/MIUCC52538.2021.9447673"},{"key":"10468_CR50","doi-asserted-by":"publisher","unstructured":"Encapera A, Gosavi A (2017) A new reinforcement learning algorithm with fixed exploration for semi-markov control in preventive maintenance. In: ASME 2017 12th international manufacturing science and engineering conference (MSEC 2017) collocated with the JSME\/ASME 2017 6th international conference on materials and processing 3. https:\/\/doi.org\/10.1115\/MSEC2017-2880","DOI":"10.1115\/MSEC2017-2880"},{"key":"10468_CR51","doi-asserted-by":"publisher","first-page":"421","DOI":"10.1016\/j.cirp.2020.04.008","volume":"69","author":"B Epureanu","year":"2020","unstructured":"Epureanu B, Li X, Nassehi A et al (2020) Self-repair of smart manufacturing systems by deep reinforcement learning. CIRP Ann 69:421\u2013424. https:\/\/doi.org\/10.1016\/j.cirp.2020.04.008","journal-title":"CIRP Annals"},{"key":"10468_CR52","doi-asserted-by":"publisher","first-page":"64","DOI":"10.1016\/j.inffus.2020.10.001","volume":"67","author":"L Erhan","year":"2021","unstructured":"Erhan L, Ndubuaku M, Di Mauro M et al (2021) Smart anomaly detection in sensor systems: a multi-perspective review. In Fusion 67:64\u201379. https:\/\/doi.org\/10.1016\/j.inffus.2020.10.001","journal-title":"Information Fusion"},{"key":"10468_CR53","unstructured":"Ericsson (2021) IoT connections outlook. https:\/\/www.ericsson.com\/en\/reports-and-papers\/mobility-report\/dataforecasts\/iot-connections-outlook"},{"key":"10468_CR54","unstructured":"Fei Y, Yang Z, Wang Z (2021) Risk-sensitive reinforcement learning with function approximation: a debiasing approach. In: International conference on machine learning (PMLR), pp 3198\u20133207"},{"key":"10468_CR55","doi-asserted-by":"publisher","unstructured":"Feng M, Li Y (2022) Predictive maintenance decision making based on reinforcement learning in multistage production systems. IEEE Access 10:18910\u201318921. https:\/\/doi.org\/10.1109\/ACCESS.2022.3151170","DOI":"10.1109\/ACCESS.2022.3151170"},{"key":"10468_CR56","doi-asserted-by":"publisher","unstructured":"Fink O, Wang Q, Svens\u00e9n M et al (2020) Potential, challenges and future directions for deep learning in prognostics and health management applications. Eng Appl Artif Intell. https:\/\/doi.org\/10.1016\/j.engappai.2020.103678","DOI":"10.1016\/j.engappai.2020.103678"},{"key":"10468_CR57","unstructured":"Fons E, Dawson P, Zeng X et\u00a0al (2021) Adaptive weighting scheme for automatic time-series data augmentation. arXiv preprint. arXiv:2102.08310"},{"issue":"10","key":"10468_CR58","doi-asserted-by":"publisher","first-page":"1390","DOI":"10.1061\/(ASCE)0733-9445(1997)123:10(1390)","volume":"123","author":"DM Frangopol","year":"1997","unstructured":"Frangopol DM, Lin KY, Estes AC (1997) Life-cycle cost design of deteriorating structures. J Struct Eng 123(10):1390\u20131401","journal-title":"Journal of structural engineering"},{"key":"10468_CR59","unstructured":"Fujimoto S, Meger D, Precup D et\u00a0al (2022) Why should I trust you, bellman? the bellman error is a poor replacement for value error. arXiv preprint. arXiv:2201.12417"},{"issue":"1","key":"10468_CR60","doi-asserted-by":"publisher","first-page":"5","DOI":"10.1023\/B:MACH.0000019802.64038.6c","volume":"55","author":"A Gosavi","year":"2004","unstructured":"Gosavi A (2004a) A reinforcement learning algorithm based on policy iteration for average reward: empirical results with yield management and convergence analysis. Mach Learn 55(1):5\u201329","journal-title":"Machine Learning"},{"issue":"3","key":"10468_CR61","doi-asserted-by":"publisher","first-page":"654","DOI":"10.1016\/S0377-2217(02)00874-3","volume":"155","author":"A Gosavi","year":"2004","unstructured":"Gosavi A (2004b) Reinforcement learning for long-run average cost. Eur J Oper Res 155(3):654\u2013674. https:\/\/doi.org\/10.1016\/S0377-2217(02)00874-3","journal-title":"European Journal of Operational Research"},{"issue":"3","key":"10468_CR62","doi-asserted-by":"publisher","first-page":"235","DOI":"10.1007\/s11633-016-1005-3","volume":"13","author":"A Gosavi","year":"2016","unstructured":"Gosavi A, Parulekar A (2016) Solving markov decision processes with downside risk adjustment. Int J Automat Comput 13(3):235\u2013245. https:\/\/doi.org\/10.1007\/s11633-016-1005-3","journal-title":"International Journal of Automation and Computing"},{"key":"10468_CR63","unstructured":"Grzes M (2017) Reward shaping in episodic reinforcement learning. In: Proceedings of the international joint conference on autonomous agents and multiagent systems (AAMAS) 1"},{"key":"10468_CR64","unstructured":"Hardt M, Recht B, Singer Y (2015) Train faster, generalize better: stability of stochastic gradient descent. In: Proceedings of the 33rd international conference on machine learning"},{"key":"10468_CR65","doi-asserted-by":"crossref","unstructured":"Hasselt HV, Guez A, Silver D (2016) Deep reinforcement learning with double Q-learning. In: 30th AAAI conference on artificial intelligence (AAAI 2016)","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"10468_CR66","doi-asserted-by":"crossref","unstructured":"Henderson P, Islam R, Bachman P et\u00a0al (2018) Deep reinforcement learning that matters. In: Proceedings of the AAAI conference on artificial intelligence, vol 32(1)","DOI":"10.1609\/aaai.v32i1.11694"},{"key":"10468_CR67","doi-asserted-by":"publisher","unstructured":"Hofmann P, Tashman Z (2020) Hidden markov models and their application for predicting failure events. Lecture notes in computer science (including subseries Lecture notes in artificial intelligence and Lecture notes in bioinformatics), vol 12139. LNCS, pp 464\u2013477. https:\/\/doi.org\/10.1007\/978-3-030-50420-5_35","DOI":"10.1007\/978-3-030-50420-5_35"},{"key":"10468_CR68","unstructured":"Hoffmann C, Alten\u00fcller T, May MC et\u00a0al (2021) Simulative dispatching optimization of maintenance resources in a semiconductor use-case using reinforcement learning. In: Simulation in Produktion und Logistik 2021, Erlangen, 15\u201317 September 2021, p 357"},{"key":"10468_CR69","doi-asserted-by":"publisher","unstructured":"Hoong\u00a0Ong K, Niyato D, Yuen C (2020) Predictive maintenance for edge-based sensor networks: a deep reinforcement learning approach. In: IEEE world forum on Internet of Things (WF-IoT 2020)\u2014symposium proceedings. https:\/\/doi.org\/10.1109\/WF-IoT48130.2020.9221098","DOI":"10.1109\/WF-IoT48130.2020.9221098"},{"key":"10468_CR70","doi-asserted-by":"publisher","DOI":"10.1002\/int.22709","author":"A Hosseinloo","year":"2021","unstructured":"Hosseinloo A, Dahleh M (2021) Deterministic policy gradient algorithms for semi-Markov decision processes. Int J Intell Syst. https:\/\/doi.org\/10.1002\/int.22709","journal-title":"International Journal of Intelligent Systems"},{"issue":"2","key":"10468_CR71","doi-asserted-by":"publisher","first-page":"181","DOI":"10.1080\/1055678031000111803","volume":"18","author":"Q Hu","year":"2003","unstructured":"Hu Q, Yue W (2003) Optimal replacement of a system according TOA semi-markov decision process in a semi-Markov environment. Optim Methods Softw 18(2):181\u2013196","journal-title":"Optimization Methods and Software"},{"key":"10468_CR72","doi-asserted-by":"publisher","unstructured":"Hu Y, Miao X, Zhang J et al (2021a) Reinforcement learning-driven maintenance strategy: a novel solution for long-term aircraft maintenance decision optimization. Comput Ind Eng. https:\/\/doi.org\/10.1016\/j.cie.2020.107056","DOI":"10.1016\/j.cie.2020.107056"},{"key":"10468_CR73","doi-asserted-by":"crossref","unstructured":"Hua Y, Wang X, Jin B et\u00a0al (2021b) HMRL: hyper-meta learning for sparse reward reinforcement learning problem. In: Proceedings of the 27th ACM SIGKDD conference on knowledge discovery & data mining, pp 637\u2013645","DOI":"10.1145\/3447548.3467242"},{"key":"10468_CR74","doi-asserted-by":"publisher","unstructured":"Huang J, Chang Q, Chakraborty N (2019) Machine preventive replacement policy for serial production lines based on reinforcement learning. In: IEEE international conference on automation science and engineering 2019, August, pp 523\u2013528. https:\/\/doi.org\/10.1109\/COASE.2019.8843338","DOI":"10.1109\/COASE.2019.8843338"},{"key":"10468_CR75","doi-asserted-by":"publisher","unstructured":"Huang J, Chang Q, Arinez J (2020) Deep reinforcement learning based preventive maintenance policy for serial production lines. Expert Syst Appl. https:\/\/doi.org\/10.1016\/j.eswa.2020.113701","DOI":"10.1016\/j.eswa.2020.113701"},{"key":"10468_CR76","unstructured":"Hui J (2021) Reinforcement learning algorithms comparison. https:\/\/jonathan-hui.medium.com\/rl-reinforcement-learning-algorithms-comparison-76df90f180cf"},{"issue":"1","key":"10468_CR77","doi-asserted-by":"publisher","first-page":"172","DOI":"10.3390\/make4010009","volume":"4","author":"M Hutsebaut-Buysse","year":"2022","unstructured":"Hutsebaut-Buysse M, Mets K, Latr\u00e9 S (2022) Hierarchical reinforcement learning: a survey and open research challenges. Mach Learn Knowl Extr 4(1):172\u2013221","journal-title":"Machine Learning and Knowledge Extraction"},{"key":"10468_CR78","doi-asserted-by":"publisher","first-page":"173","DOI":"10.1613\/jair.1.12440","volume":"73","author":"RT Icarte","year":"2022","unstructured":"Icarte RT, Klassen TQ, Valenzano R et al (2022) Reward machines: exploiting reward function structure in reinforcement learning. J Artif Intell Res 73:173\u2013208","journal-title":"Journal of Artificial Intelligence Research"},{"key":"10468_CR79","doi-asserted-by":"publisher","first-page":"49,494","DOI":"10.1109\/ACCESS.2022.3170582","volume":"10","author":"T Imagawa","year":"2022","unstructured":"Imagawa T, Hiraoka T, Tsuruoka Y (2022) Off-policy meta-reinforcement learning with belief-based task inference. IEEE Access 10:49494\u201349507","journal-title":"IEEE Access"},{"key":"10468_CR80","unstructured":"Jaakkola T, Singh S, Jordan M (1994) Reinforcement learning algorithm for partially observable markov decision problems. Adv Neural Inf Process Syst. https:\/\/proceedings.neurips.cc\/paper\/1994\/file\/1c1d4df596d01da60385f0bb17a4a9e0-Paper.pdf"},{"key":"10468_CR81","doi-asserted-by":"publisher","unstructured":"Jha M, Theilliol D, Biswas G et\u00a0al (2019a) Approximate q-learning approach for health aware control design. In: Conference on control and fault-tolerant systems (SysTol), pp 418\u2013423. https:\/\/doi.org\/10.1109\/SYSTOL.2019.8864756","DOI":"10.1109\/SYSTOL.2019.8864756"},{"key":"10468_CR82","doi-asserted-by":"publisher","unstructured":"Jha M, Weber P, Theilliol D et\u00a0al (2019b) A reinforcement learning approach to health aware control strategy. In: 27th Mediterranean conference on control and automation (MED 2019)\u2014proceedings, pp 171\u2013176. https:\/\/doi.org\/10.1109\/MED.2019.8798548","DOI":"10.1109\/MED.2019.8798548"},{"key":"10468_CR83","doi-asserted-by":"publisher","unstructured":"Kabir F, Foggo B, Yu N (2018) Data driven predictive maintenance of distribution transformers. In: 2018 China international conference on electricity distribution (CICED), pp 312\u2013316. https:\/\/doi.org\/10.1109\/CICED.2018.8592417","DOI":"10.1109\/CICED.2018.8592417"},{"key":"10468_CR84","doi-asserted-by":"publisher","first-page":"13","DOI":"10.1016\/j.arcontrol.2020.08.003","volume":"50","author":"S Khan","year":"2020","unstructured":"Khan S, Farnsworth M, McWilliam R et al (2020) On the requirements of digital twin-driven autonomous maintenance. Annu Rev Control 50:13\u201328. https:\/\/doi.org\/10.1016\/j.arcontrol.2020.08.003","journal-title":"Annual Reviews in Control"},{"key":"10468_CR85","doi-asserted-by":"publisher","unstructured":"Knowles M, Baglee D, Wermter S (2011) Reinforcement learning for scheduling of maintenance. In: Research and development in intelligent systems XXVII: incorporating applications and innovations in Intelligent systems XVIII\u2014AI 2010, 30th SGAI international conference on innovative techniques and applications of artificial intelligence, pp 409\u2013422. https:\/\/doi.org\/10.1007\/978-0-85729-130-1_31","DOI":"10.1007\/978-0-85729-130-1_31"},{"issue":"2","key":"10468_CR86","doi-asserted-by":"publisher","first-page":"231","DOI":"10.3390\/electronics8020231","volume":"8","author":"P Kofinas","year":"2019","unstructured":"Kofinas P, Dounis AI (2019) Online tuning of a PID controller with a fuzzy reinforcement learning mas for flow rate control of a desalination unit. Electronics 8(2):231","journal-title":"Electronics"},{"issue":"1","key":"10468_CR87","doi-asserted-by":"publisher","first-page":"33","DOI":"10.1007\/s11740-018-0855-7","volume":"13","author":"A Kuhnle","year":"2019","unstructured":"Kuhnle A, Jakubik J, Lanza G (2019) Reinforcement learning for opportunistic maintenance optimization. Prod Eng 13(1):33\u201341","journal-title":"Production Engineering"},{"key":"10468_CR88","unstructured":"Laape S, Dollar B, Cotteleer M et\u00a0al (2020) Implementing the smart factory. Deloitte Insights. https:\/\/www2.deloitte.com\/us\/en\/insights\/topics\/digital-transformation\/smart-factory-2-0-technology-initiatives.html"},{"key":"10468_CR89","doi-asserted-by":"crossref","unstructured":"Lange S, Gabel T, Riedmiller M (2012) Batch reinforcement learning, reinforcement learning. In: Wiering M, van Otterlo M (eds) Reinforcement learning. Adaptation, learning, and optimization. Springer, Berlin, pp 45\u201373","DOI":"10.1007\/978-3-642-27645-3_2"},{"issue":"1\u20132","key":"10468_CR90","doi-asserted-by":"publisher","first-page":"314","DOI":"10.1016\/j.ymssp.2013.06.004","volume":"42","author":"J Lee","year":"2014","unstructured":"Lee J, Wu F, Zhao W et al (2014) Prognostics and health management design for rotary machinery systems\u2014reviews, methodology and applications. Mech Syst Signal Process 42(1\u20132):314\u2013334","journal-title":"Mechanical systems and signal processing"},{"key":"10468_CR91","doi-asserted-by":"publisher","unstructured":"Lepenioti K, Pertselakis M, Bousdekis A et\u00a0al (2020) Machine learning for predictive and prescriptive analytics of operational data in smart manufacturing. Lecture notes in business information processing, vol 382 LNBIP, pp 5\u201316. https:\/\/doi.org\/10.1007\/978-3-030-49165-9_1","DOI":"10.1007\/978-3-030-49165-9_1"},{"key":"10468_CR92","unstructured":"Lewis F, Vrabie D, Vamvoudakis K (2012) Reinforcement learning and feedback control: Using natural decision methods to design optimal adaptive controllers. https:\/\/ieeexplore.ieee.org\/document\/6315769"},{"key":"10468_CR93","doi-asserted-by":"publisher","unstructured":"Li Z (2019) CWRU bearing dataset and Gearbox dataset of IEEE PHM challenge competition in 2009. https:\/\/doi.org\/10.21227\/g8ts-zd15","DOI":"10.21227\/g8ts-zd15"},{"key":"10468_CR94","doi-asserted-by":"publisher","unstructured":"Li Z, Guo J, Zhou R (2016) Maintenance scheduling optimization based on reliability and prognostics information. In: 2016 Annual reliability and maintainability symposium (RAMS), pp 1\u20135. https:\/\/doi.org\/10.1109\/RAMS.2016.7448069","DOI":"10.1109\/RAMS.2016.7448069"},{"key":"10468_CR95","doi-asserted-by":"crossref","unstructured":"Li B, Zhou Y (2020) Multi-component maintenance optimization: an approach combining genetic algorithm and multiagent reinforcement learning. In: 2020 global reliability and prognostics and health management (PHM\u2014Shanghai), pp 1\u20137","DOI":"10.1109\/PHM-Shanghai49105.2020.9280997"},{"issue":"14","key":"10468_CR96","doi-asserted-by":"publisher","first-page":"3823","DOI":"10.1080\/00207540701829752","volume":"47","author":"J Li","year":"2009","unstructured":"Li J, Blumenfeld DE, Huang N et al (2009) Throughput analysis of production systems: recent advances and future topics. Int J Prod Res 47(14):3823\u20133851. https:\/\/doi.org\/10.1080\/00207540701829752","journal-title":"International Journal of Production Research"},{"issue":"149","key":"10468_CR97","first-page":"562","volume":"5","author":"X Li","year":"2013","unstructured":"Li X, Qian J, Gg Wang (2013) Fault prognostic based on hybrid method of state judgment and regression. Adv Mech Eng 5(149):562","journal-title":"Advances in Mechanical Engineering"},{"issue":"9","key":"10468_CR98","doi-asserted-by":"publisher","first-page":"2133","DOI":"10.1016\/j.cja.2019.07.003","volume":"32","author":"Z Li","year":"2019","unstructured":"Li Z, Zhong S, Lin L (2019) An aero-engine life-cycle maintenance policy optimization algorithm: reinforcement learning approach. Chin J Aeronaut 32(9):2133\u20132150. https:\/\/doi.org\/10.1016\/j.cja.2019.07.003","journal-title":"Chinese Journal of Aeronautics"},{"key":"10468_CR99","doi-asserted-by":"publisher","unstructured":"Li L, Liu J, Wei S et al (2021) Smart robot-enabled remaining useful life prediction and maintenance optimization for complex structures using artificial intelligence and machine learning. Proc SPIE. https:\/\/doi.org\/10.1117\/12.2589045","DOI":"10.1117\/12.2589045"},{"key":"10468_CR100","unstructured":"Lillicrap TP, Hunt JJ, Pritzel A et\u00a0al (2015) Continuous control with deep reinforcement learning. arXiv e-prints. arXiv:1509.02971"},{"key":"10468_CR101","doi-asserted-by":"publisher","unstructured":"Ling Z, Wang X, Qu F (2018) Reinforcement learning-based maintenance scheduling for resource constrained flow line system. In: 2018 IEEE 4th international conference on control science and systems engineering (ICCSSE 2018), pp 364\u2013369. https:\/\/doi.org\/10.1109\/CCSSE.2018.8724807","DOI":"10.1109\/CCSSE.2018.8724807"},{"issue":"3","key":"10468_CR102","doi-asserted-by":"publisher","first-page":"652","DOI":"10.1109\/TASE.2013.2250282","volume":"10","author":"K Liu","year":"2013","unstructured":"Liu K, Gebraeel NZ, Shi J (2013) A data-level fusion model for developing composite health indices for degradation modeling and prognostic analysis. IEEE Trans Automat Sci Eng 10(3):652\u2013664","journal-title":"IEEE Transactions on Automation Science and Engineering"},{"issue":"1","key":"10468_CR103","doi-asserted-by":"publisher","first-page":"299","DOI":"10.1109\/TASE.2016.2517155","volume":"14","author":"L Liu","year":"2017","unstructured":"Liu L, Wang Z, Zhang H (2017) Adaptive fault-tolerant tracking control for MIMO discrete-time systems via reinforcement learning algorithm with less learning parameters. IEEE Trans Automat Sci Eng 14(1):299\u2013313. https:\/\/doi.org\/10.1109\/TASE.2016.2517155","journal-title":"IEEE Transactions on Automation Science and Engineering"},{"issue":"1","key":"10468_CR104","doi-asserted-by":"publisher","first-page":"166","DOI":"10.1016\/j.ejor.2019.10.049","volume":"283","author":"Y Liu","year":"2020","unstructured":"Liu Y, Chen Y, Jiang T (2020) Dynamic selective maintenance optimization for multi-state systems over a finite horizon: a deep reinforcement learning approach. Eur J Oper Res 283(1):166\u2013181. https:\/\/doi.org\/10.1016\/j.ejor.2019.10.049","journal-title":"European Journal of Operational Research"},{"key":"10468_CR105","doi-asserted-by":"publisher","unstructured":"Luo Y (2021) Application of reinforcement learning algorithm model in gas path fault intelligent diagnosis of gas turbine. Comput Intell Neurosci. https:\/\/doi.org\/10.1155\/2021\/3897077","DOI":"10.1155\/2021\/3897077"},{"key":"10468_CR106","doi-asserted-by":"crossref","unstructured":"Ma Z, Guo J, Mao S et\u00a0al (2020) An interpretability research of the XGBoost algorithm in remaining useful life prediction. In: 2020 International conference on big data & artificial intelligence & software engineering (ICBASE), pp 433\u2013438","DOI":"10.1109\/ICBASE51474.2020.00098"},{"key":"10468_CR107","doi-asserted-by":"publisher","first-page":"111","DOI":"10.1016\/j.enbuild.2017.05.055","volume":"150","author":"K Macek","year":"2017","unstructured":"Macek K, Endel P, Cauchi N et al (2017) Long-term predictive maintenance: a study of optimal cleaning of biomass boilers. Energy Build 150:111\u2013117","journal-title":"Energy and Buildings"},{"key":"10468_CR108","unstructured":"Mahadevan S, Marchalleck N, Das TK et\u00a0al (1997) Self-improving factory simulation using continuous-time average-reward reinforcement learning. In: Machine learning international workshop. Morgan Kaufmann Publishers, Los Angeles"},{"key":"10468_CR109","doi-asserted-by":"crossref","unstructured":"Mahmood AR, Sutton RS, Degris T et\u00a0al (2012) Tuning-free step-size adaptation. In: 2012 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 2121\u20132124","DOI":"10.1109\/ICASSP.2012.6288330"},{"key":"10468_CR110","unstructured":"Mann L, Saxena A, Knapp GM (1995) Statistical-based or condition-based preventive maintenance? J Qual Maintenance Eng 6(5):519\u2013541"},{"key":"10468_CR111","doi-asserted-by":"publisher","DOI":"10.1007\/s00170-021-08290-x","author":"H Mao","year":"2021","unstructured":"Mao H, Liu Z, Qiu C (2021) Adaptive disassembly sequence planning for VR maintenance training via deep reinforcement learning. Int J Adv Manuf Technol. https:\/\/doi.org\/10.1007\/s00170-021-08290-x","journal-title":"International Journal of Advanced Manufacturing Technology"},{"key":"10468_CR112","doi-asserted-by":"publisher","unstructured":"Martinez C, Perrin G, Ramasso E et\u00a0al (2018) A deep reinforcement learning approach for early classification of time series. In: European signal processing conference 2018, September, pp 2030\u20132034. https:\/\/doi.org\/10.23919\/EUSIPCO.2018.8553544","DOI":"10.23919\/EUSIPCO.2018.8553544"},{"key":"10468_CR113","doi-asserted-by":"publisher","unstructured":"Mattioli J, Perico P, Robic PO (2020) Improve total production maintenance with artificial intelligence. In: Proceedings\u20142020 3rd international conference on artificial intelligence for industries (AI4I 2020), pp 56\u201359. https:\/\/doi.org\/10.1109\/AI4I49448.2020.00019","DOI":"10.1109\/AI4I49448.2020.00019"},{"key":"10468_CR114","doi-asserted-by":"crossref","unstructured":"Mehndiratta M, Camci E, Kayacan E (2018) Automated tuning of nonlinear model predictive controller by reinforcement learning. In: 2018 IEEE\/RSJ international conference on intelligent robots and systems (IROS), pp 3016\u20133021","DOI":"10.1109\/IROS.2018.8594350"},{"key":"10468_CR115","doi-asserted-by":"publisher","first-page":"443","DOI":"10.1016\/0043-1648(95)90158-2","volume":"181","author":"H Meng","year":"1995","unstructured":"Meng H, Ludema K (1995) Wear models and predictive equations: their form and content. Wear 181:443\u2013457","journal-title":"Wear"},{"key":"10468_CR116","unstructured":"Mikhail M, Yacout S, Ouali M (2019) Optimal preventive maintenance strategy using reinforcement learning. In: Proceedings of the international conference on industrial engineering and operations management, pp 133\u2013141"},{"key":"10468_CR117","doi-asserted-by":"publisher","unstructured":"Min W, Chao Q (2012) Reinforcement learning based maintenance scheduling for a two-machine flow line with deteriorating quality states. In: Proceedings\u20142012 3rd global congress on intelligent systems (GCIS 2012), pp 176\u2013179. https:\/\/doi.org\/10.1109\/GCIS.2012.82","DOI":"10.1109\/GCIS.2012.82"},{"issue":"1","key":"10468_CR118","doi-asserted-by":"publisher","first-page":"276","DOI":"10.3390\/make4010013","volume":"4","author":"J Moos","year":"2022","unstructured":"Moos J, Hansel K, Abdulsamad H et al (2022) Robust reinforcement learning: a review of foundations and recent advances. Mach Learn Knowl Extr 4(1):276\u2013315","journal-title":"Machine Learning and Knowledge Extraction"},{"issue":"2","key":"10468_CR119","doi-asserted-by":"publisher","first-page":"335","DOI":"10.1162\/0899766053011528","volume":"17","author":"J Morimoto","year":"2005","unstructured":"Morimoto J, Doya K (2005) Robust reinforcement learning. Neural Comput 17(2):335\u2013359","journal-title":"Neural computation"},{"key":"10468_CR120","unstructured":"Nair A, Gupta A, Dalal M et\u00a0al (2020) AWAC: accelerating online reinforcement learning with offline datasets. arXiv preprint. arXiv:2006.09359"},{"key":"10468_CR121","unstructured":"Narvekar S, Peng B, Leonetti M et\u00a0al (2020) Curriculum learning for reinforcement learning domains: a framework and survey. CoRR. arXiv:2003.04960"},{"key":"10468_CR122","unstructured":"Nectoux P, Gouriveau R, Medjaher K et\u00a0al (2012) Pronostia: an experimental platform for bearings accelerated degradation tests. In: IEEE international conference on prognostics and health management (PHM\u201912), pp 1\u20138"},{"key":"10468_CR123","doi-asserted-by":"crossref","unstructured":"Ng AY, Coates A, Diel M et\u00a0al (2006) Autonomous inverted helicopter flight via reinforcement learning. Experimental Robotics IX pp 363\u2013372","DOI":"10.1007\/11552246_35"},{"key":"10468_CR124","doi-asserted-by":"publisher","unstructured":"Ong K, Wenbo W, Friedrichs T et\u00a0al (2021a) Augmented human intelligence for decision making in maintenance risk taking tasks using reinforcement learning. In: Conference proceedings\u2014IEEE international conference on systems, man and cybernetics, pp 3114\u20133120. https:\/\/doi.org\/10.1109\/SMC52423.2021.9658936","DOI":"10.1109\/SMC52423.2021.9658936"},{"key":"10468_CR125","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2021.3109955","author":"K Ong","year":"2021","unstructured":"Ong K, Wenbo W, Niyato D et al (2021b) Deep reinforcement learning based predictive maintenance model for effective resource management in industrial IoT. IEEE Internet Things J. https:\/\/doi.org\/10.1109\/JIOT.2021.3109955","journal-title":"IEEE Internet of Things Journal"},{"issue":"9","key":"10468_CR126","doi-asserted-by":"publisher","first-page":"2317","DOI":"10.3390\/en11092317","volume":"11","author":"S Ozturk","year":"2018","unstructured":"Ozturk S, Fthenakis V, Faulstich S (2018) Failure modes, effects and criticality analysis for wind turbines considering climatic regions and comparing geared and direct drive wind turbines. Energies 11(9):2317","journal-title":"Energies"},{"key":"10468_CR127","doi-asserted-by":"crossref","unstructured":"Panzer M, Bender B (2021) Deep reinforcement learning in production systems: a systematic literature review. Int J Prod Res 60(3):1\u201326","DOI":"10.1080\/00207543.2021.1973138"},{"key":"10468_CR128","doi-asserted-by":"publisher","first-page":"470","DOI":"10.1016\/j.jmsy.2020.07.004","volume":"56","author":"P Paraschos","year":"2020","unstructured":"Paraschos P, Koulinas G, Koulouriotis D (2020) Reinforcement learning for combined production-maintenance and quality control of a manufacturing system with deterioration failures. J Manuf Syst 56:470\u2013483. https:\/\/doi.org\/10.1016\/j.jmsy.2020.07.004","journal-title":"Journal of Manufacturing Systems"},{"key":"10468_CR129","unstructured":"Patil S, Abbeel P (2013) Partially observable markov decision processes (POMDPs). Guest Lecture: CS287 advanced robotics"},{"key":"10468_CR130","doi-asserted-by":"publisher","unstructured":"Pinciroli L, Baraldi P, Compare M et\u00a0al (2020) Agent-based modeling and reinforcement learning for optimizing energy systems operation and maintenance: the pathmind solution. In: Proceedings of the 30th European safety and reliability conference and the 15th probabilistic safety assessment and management conference, pp 1476\u20131480. https:\/\/doi.org\/10.3850\/978-981-14-8593-0_5863-cd","DOI":"10.3850\/978-981-14-8593-0_5863-cd"},{"key":"10468_CR131","doi-asserted-by":"publisher","unstructured":"Pinciroli L, Baraldi P, Ballabio G et\u00a0al (2021) Deep reinforcement learning based on proximal policy optimization for the maintenance of a wind farm with multiple crews. Energies. https:\/\/doi.org\/10.3390\/en14206743","DOI":"10.3390\/en14206743"},{"key":"10468_CR132","doi-asserted-by":"publisher","first-page":"752","DOI":"10.1016\/j.renene.2021.11.052","volume":"183","author":"L Pinciroli","year":"2022","unstructured":"Pinciroli L, Baraldi P, Ballabio G et al (2022) Optimization of the operation and maintenance of renewable energy systems by deep reinforcement learning. Renew Energy 183:752\u2013763. https:\/\/doi.org\/10.1016\/j.renene.2021.11.052","journal-title":"Renewable Energy"},{"key":"10468_CR133","unstructured":"Pinto L, Davidson J, Sukthankar R et\u00a0al (2017a) Robust adversarial reinforcement learning. In: International conference on machine learning (PMLR), pp 2817\u20132826"},{"key":"10468_CR134","unstructured":"Plappert M, Houthooft R, Dhariwal P et\u00a0al (2017b) Parameter space noise for exploration. arXiv preprint. arXiv:1706.01905"},{"issue":"3","key":"10468_CR135","doi-asserted-by":"publisher","first-page":"239","DOI":"10.1002\/nav.20347","volume":"56","author":"WB Powell","year":"2009","unstructured":"Powell WB (2009) What you should know about approximate dynamic programming. Naval Res Logist NRL) 56(3):239\u2013249","journal-title":"Naval Research Logistics (NRL)"},{"issue":"5","key":"10468_CR136","doi-asserted-by":"publisher","first-page":"537","DOI":"10.1561\/2200000091","volume":"15","author":"L Prashanth","year":"2022","unstructured":"Prashanth L, Fu MC et al (2022) Risk-sensitive reinforcement learning via policy gradient search. Found Trends Mach Learn 15(5):537\u2013693","journal-title":"Foundations and Trends\u00ae in Machine Learning"},{"key":"10468_CR137","unstructured":"Prognostics HM Society (2010) 2010 PHM society conference data challenge. https:\/\/phmsociety.org\/phm_competition\/2010-phm-society-conference-data-challenge\/"},{"key":"10468_CR138","doi-asserted-by":"crossref","unstructured":"Ramasso E (2014) Investigating computational geometry for failure prognostics in presence of imprecise health indicator: results and comparisons on C-MAPSS datasets. In: PHM society European conference 2(1)","DOI":"10.36001\/phme.2014.v2i1.1460"},{"key":"10468_CR139","doi-asserted-by":"publisher","unstructured":"Ren Y (2021) Optimizing predictive maintenance with machine learning for reliability improvement. ASCE ASME J Risk Uncertain Eng Syst Part B Mech Eng. https:\/\/doi.org\/10.1115\/1.4049525","DOI":"10.1115\/1.4049525"},{"key":"10468_CR140","doi-asserted-by":"publisher","first-page":"291","DOI":"10.1016\/j.apenergy.2019.03.027","volume":"241","author":"R Rocchetta","year":"2019","unstructured":"Rocchetta R, Bellani L, Compare M et al (2019) A reinforcement learning framework for optimal operation and maintenance of power grids. Appl Energy 241:291\u2013301. https:\/\/doi.org\/10.1016\/j.apenergy.2019.03.027","journal-title":"Applied Energy"},{"key":"10468_CR141","unstructured":"Russenschuck S (1999) Mathematical optimization techniques. Tech. rep., CERN"},{"key":"10468_CR142","doi-asserted-by":"crossref","unstructured":"Sateesh\u00a0Babu G, Zhao P, Li XL (2016) Deep convolutional neural network based regression approach for estimation of remaining useful life. In: International conference on database systems for advanced applications, pp 214\u2013228","DOI":"10.1007\/978-3-319-32025-0_14"},{"key":"10468_CR143","unstructured":"Saxena A, Goebel K (2008) Turbofan engine degradation simulation data set. http:\/\/ti.arc.nasa.gov\/project\/prognostic-data-repository"},{"key":"10468_CR144","doi-asserted-by":"crossref","unstructured":"Saxena A, Goebel K, Simon D et\u00a0al (2008) Damage propagation modeling for aircraft engine run-to-failure simulation. In: 2008 international conference on prognostics and health management, pp 1\u20139","DOI":"10.1109\/PHM.2008.4711414"},{"key":"10468_CR145","doi-asserted-by":"crossref","unstructured":"Saxena A, Celaya J, Saha B et\u00a0al (2010a) Evaluating prognostics performance for algorithms incorporating uncertainty estimates. In: 2010 IEEE aerospace conference, pp 1\u201311","DOI":"10.1109\/AERO.2010.5446828"},{"issue":"1","key":"10468_CR146","first-page":"4","volume":"1","author":"A Saxena","year":"2010","unstructured":"Saxena A, Celaya J, Saha B et al (2010b) Metrics for offline evaluation of prognostic performance. Int J Prognost Health Manag 1(1):4\u201323","journal-title":"International Journal of Prognostics and health management"},{"issue":"4","key":"10468_CR147","doi-asserted-by":"publisher","first-page":"04014,120","DOI":"10.1061\/(ASCE)ST.1943-541X.0001038","volume":"141","author":"D Saydam","year":"2015","unstructured":"Saydam D, Frangopol DM (2015) Risk-based maintenance optimization of deteriorating bridges. J Struct Eng 141(4):04014120. https:\/\/doi.org\/10.1061\/(ASCE)ST.1943-541X.0001038","journal-title":"Journal of Structural Engineering"},{"issue":"9","key":"10468_CR148","doi-asserted-by":"publisher","first-page":"6611","DOI":"10.1007\/s00170-022-09784-y","volume":"121","author":"S Sayyad","year":"2022","unstructured":"Sayyad S, Kumar S, Bongale A et al (2022) Tool wear prediction using long short-term memory variants and hybrid feature selection techniques. Int J Adv Manuf Technol 121(9):6611\u20136633","journal-title":"The International Journal of Advanced Manufacturing Technology"},{"key":"10468_CR149","doi-asserted-by":"crossref","unstructured":"Schaefer AM, Udluft S, Zimmermann HG (2007) A recurrent control neural network for data efficient reinforcement learning. In: 2007 IEEE international symposium on approximate dynamic programming and reinforcement learning (IEEE), pp 151\u2013157","DOI":"10.1109\/ADPRL.2007.368182"},{"issue":"3","key":"10468_CR150","first-page":"161","volume":"41","author":"P Scheibelhofer","year":"2012","unstructured":"Scheibelhofer P, Gleispach D, Hayderer G et al (2012) A methodology for predictive maintenance in semiconductor manufacturing. Aust J Stat 41(3):161\u2013173","journal-title":"Austrian Journal of Statistics"},{"key":"10468_CR151","doi-asserted-by":"publisher","unstructured":"Senthil C, Pandian R (2022) Proactive maintenance model using reinforcement learning algorithm in rubber industry. Processes. https:\/\/doi.org\/10.3390\/pr10020371","DOI":"10.3390\/pr10020371"},{"issue":"7","key":"10468_CR152","doi-asserted-by":"publisher","first-page":"1298","DOI":"10.1162\/NECO_a_00600","volume":"26","author":"Y Shen","year":"2014","unstructured":"Shen Y, Tobia MJ, Sommer T et al (2014) Risk-sensitive reinforcement learning. Neural Comput 26(7):1298\u20131328","journal-title":"Neural computation"},{"key":"10468_CR153","doi-asserted-by":"publisher","unstructured":"Shi Y, Xiang Y, Jin T (2019) Structured maintenance policies for deteriorating transportation infrastructures: combination of maintenance types. In: Proceedings of annual reliability and maintainability symposium 2019, January. https:\/\/doi.org\/10.1109\/RAMS.2019.8769227","DOI":"10.1109\/RAMS.2019.8769227"},{"key":"10468_CR154","doi-asserted-by":"publisher","first-page":"183","DOI":"10.1016\/j.neucom.2020.03.063","volume":"402","author":"Q Shi","year":"2020","unstructured":"Shi Q, Lam HK, Xuan C et al (2020) Adaptive neuro-fuzzy pid controller based on twin delayed deep deterministic policy gradient algorithm. Neurocomputing 402:183\u2013194. https:\/\/doi.org\/10.1016\/j.neucom.2020.03.063","journal-title":"Neurocomputing"},{"key":"10468_CR155","doi-asserted-by":"publisher","unstructured":"Shuvo S, Yilmaz Y (2020) Predictive maintenance for increasing EV charging load in distribution power system. In: 2020 IEEE international conference on communications, control, and computing technologies for smart grids, SmartGridComm 2020 https:\/\/doi.org\/10.1109\/SmartGridComm47815.2020.9303021","DOI":"10.1109\/SmartGridComm47815.2020.9303021"},{"key":"10468_CR156","first-page":"284","volume":"1994","author":"SP Singh","year":"1994","unstructured":"Singh SP, Jaakkola T, Jordan MI (1994) Learning without state-estimation in partially observable Markovian decision processes. Mach Learn Proc 1994:284\u2013292","journal-title":"Machine Learning Proceedings"},{"key":"10468_CR157","unstructured":"Sinha S (2021) State of IoT 2021. https:\/\/iot-analytics.com\/number-connected-iot-devices\/"},{"key":"10468_CR158","doi-asserted-by":"publisher","unstructured":"Skordilis E, Moghaddass R (2020) A deep reinforcement learning approach for real-time sensor-driven decision making and predictive analytics. Comput Ind Eng. https:\/\/doi.org\/10.1016\/j.cie.2020.106600","DOI":"10.1016\/j.cie.2020.106600"},{"issue":"108","key":"10468_CR159","first-page":"691","volume":"170","author":"MR Skydt","year":"2021","unstructured":"Skydt MR, Bang M, Shaker HR (2021) A probabilistic sequence classification approach for early fault prediction in distribution grids using long short-term memory neural networks. Measurement 170(108):691","journal-title":"Measurement"},{"key":"10468_CR160","unstructured":"Song X, Jiang Y, Tu S et\u00a0al (2019) Observational overfitting in reinforcement learning. arXiv preprint. arXiv:1912.02975"},{"issue":"116","key":"10468_CR161","doi-asserted-by":"publisher","first-page":"323","DOI":"10.1016\/j.eswa.2021.116323","volume":"192","author":"J Su","year":"2022","unstructured":"Su J, Huang J, Adams S et al (2022) Deep multi-agent reinforcement learning for multi-level preventive maintenance in manufacturing systems. Expert Syst Appl 192(116):323. https:\/\/doi.org\/10.1016\/j.eswa.2021.116323","journal-title":"Expert Systems with Applications"},{"key":"10468_CR162","doi-asserted-by":"crossref","unstructured":"Susto GA, Schirru A, Pampuri S et\u00a0al (2013) A predictive maintenance system for integral type faults based on support vector machines: an application to ion implantation. In: 2013 IEEE international conference on automation science and engineering (CASE), pp 195\u2013200","DOI":"10.1109\/CoASE.2013.6653952"},{"key":"10468_CR163","doi-asserted-by":"crossref","unstructured":"Susto GA, Wan J, Pampuri S et\u00a0al (2014) An adaptive machine learning decision system for flexible predictive maintenance. In: 2014 IEEE international conference on automation science and engineering (CASE), pp 806\u2013811","DOI":"10.1109\/CoASE.2014.6899418"},{"key":"10468_CR164","volume-title":"Reinforcement Learning: An Introduction","author":"R Sutton","year":"2018","unstructured":"Sutton R, Barto A (2018) Reinforcement learning: an introduction, 2nd edn. MIT, Cambridge","edition":"2"},{"issue":"1\u20132","key":"10468_CR165","doi-asserted-by":"publisher","first-page":"181","DOI":"10.1016\/S0004-3702(99)00052-1","volume":"112","author":"RS Sutton","year":"1999","unstructured":"Sutton RS, Precup D, Singh S (1999) Between mdps and semi-MDPS: a framework for temporal abstraction in reinforcement learning. Artif Intell 112(1\u20132):181\u2013211","journal-title":"Artificial intelligence"},{"key":"10468_CR166","doi-asserted-by":"crossref","unstructured":"Swazinna P, Udluft S, Hein D et\u00a0al (2022) Comparing model-free and model-based algorithms for offline reinforcement learning. arXiv preprint. arXiv:2201.05433","DOI":"10.1016\/j.ifacol.2022.07.602"},{"key":"10468_CR167","doi-asserted-by":"publisher","first-page":"46,788","DOI":"10.1109\/ACCESS.2021.3059244","volume":"9","author":"A Tanimoto","year":"2021","unstructured":"Tanimoto A (2021) Combinatorial Q-learning for condition-based infrastructure maintenance. IEEE Access 9:46788-46799. https:\/\/doi.org\/10.1109\/ACCESS.2021.3059244","journal-title":"IEEE Access"},{"issue":"1","key":"10468_CR168","first-page":"6","volume":"37","author":"M Templier","year":"2015","unstructured":"Templier M, Par\u00e9 G (2015) A framework for guiding and evaluating literature reviews. Commun Assoc Inf Syst 37(1):6","journal-title":"Communications of the Association for Information Systems"},{"key":"10468_CR169","unstructured":"Thomas D (2020) Manufacturing machinery maintenance\u2014NIST. National Institute of Standards and Technology (NIST), Gaithersburg. https:\/\/www.nist.gov\/el\/applied-economics-office\/manufacturing\/topics-manufacturing\/manufacturing-machinery-maintenance"},{"key":"10468_CR170","doi-asserted-by":"publisher","unstructured":"Thomas DS, Weiss BA (2020) Economics of manufacturing machinery maintenance. National Institute of Standards and Technology (NIST), Gaithersburg. https:\/\/doi.org\/10.6028\/NIST.AMS.100-34https:\/\/nvlpubs.nist.gov\/nistpubs\/ams\/NIST.AMS.100-34.pdf","DOI":"10.6028\/NIST.AMS.100-34"},{"key":"10468_CR171","doi-asserted-by":"publisher","first-page":"518","DOI":"10.1016\/j.jmsy.2022.07.016","volume":"64","author":"A Valet","year":"2022","unstructured":"Valet A, Altenm\u00fcller T, Waschneck B et al (2022) Opportunistic maintenance scheduling with deep reinforcement learning. J Manuf Syst 64:518\u2013534","journal-title":"Journal of Manufacturing Systems"},{"key":"10468_CR172","unstructured":"Vogl GW, Qiao H (2021) Monitoring, diagnostics and prognostics for manufacturing operations (NIST). National Institute of Standards and Technology (NIST), Gaithersburg. https:\/\/www.nist.gov\/programs-projects\/monitoring-diagnostics-and-prognostics-manufacturing-operations"},{"key":"10468_CR173","unstructured":"Walsh C (2022) Paris-Erdogan equation. https:\/\/www.maths.tcd.ie\/~chas\/node24.html#SECTION00841000000000000000"},{"issue":"1","key":"10468_CR174","doi-asserted-by":"publisher","first-page":"9","DOI":"10.12733\/jcis8124","volume":"10","author":"X Wang","year":"2014","unstructured":"Wang X, Wang H, Qi C et al (2014) Reinforcement learning based predictive maintenance for a machine with multiple deteriorating yield levels. J Comput Inf Syst 10(1):9\u201319. https:\/\/doi.org\/10.12733\/jcis8124","journal-title":"Journal of Computational Information Systems"},{"key":"10468_CR175","doi-asserted-by":"publisher","unstructured":"Wang X, Qi C, Wang H et\u00a0al (2015) Resilience-driven maintenance scheduling methodology for multi-agent production line system. In: Proceedings of the 2015 27th Chinese control and decision conference (CCDC 2015), pp 614\u2013619. https:\/\/doi.org\/10.1109\/CCDC.2015.7161844","DOI":"10.1109\/CCDC.2015.7161844"},{"issue":"2","key":"10468_CR176","doi-asserted-by":"publisher","first-page":"325","DOI":"10.1007\/s10845-013-0864-5","volume":"27","author":"X Wang","year":"2016","unstructured":"Wang X, Wang H, Qi C (2016) Multi-agent reinforcement learning based maintenance policy for a resource constrained flow line system. J Intell Manuf 27(2):325\u2013333. https:\/\/doi.org\/10.1007\/s10845-013-0864-5","journal-title":"Journal of Intelligent Manufacturing"},{"key":"10468_CR177","doi-asserted-by":"publisher","unstructured":"Wang H, Yan Q, Zhang S (2021a) Integrated scheduling and flexible maintenance in deteriorating multi-state single machine system using a reinforcement learning approach. Adv Eng Inf. https:\/\/doi.org\/10.1016\/j.aei.2021.101339","DOI":"10.1016\/j.aei.2021.101339"},{"key":"10468_CR178","doi-asserted-by":"publisher","unstructured":"Wang X, Wang Y, Dai H (2021b) Fault diagnosis based on data-driven dynamic model. In: ICSMD 2021\u20142nd international conference on sensing, measurement and data analytics in the era of artificial intelligence. https:\/\/doi.org\/10.1109\/ICSMD53520.2021.9670767","DOI":"10.1109\/ICSMD53520.2021.9670767"},{"key":"10468_CR179","doi-asserted-by":"publisher","unstructured":"Wang X, Xu D, Qu N et al (2021c) Predictive maintenance and sensitivity analysis for equipment with multiple quality states. Math Probl Eng. https:\/\/doi.org\/10.1155\/2021\/4914372","DOI":"10.1155\/2021\/4914372"},{"key":"10468_CR180","doi-asserted-by":"crossref","unstructured":"Wang X, Zhang G, Li Y et\u00a0al (2022) A heuristically accelerated reinforcement learning method for maintenance policy of an assembly line. J Ind Manag Optim 19(4):2381\u20132395","DOI":"10.3934\/jimo.2022047"},{"key":"10468_CR181","doi-asserted-by":"crossref","unstructured":"Weibull W (1951) A statistical distribution function of wide applicability. J Appl Mech 18:293\u2013297","DOI":"10.1115\/1.4010337"},{"key":"10468_CR182","doi-asserted-by":"publisher","first-page":"13","DOI":"10.1016\/J.IFACOL.2016.12.154","volume":"49","author":"BA Weiss","year":"2016","unstructured":"Weiss BA, Helu M, Vogl G et al (2016) Use case development to advance monitoring, diagnostics, and prognostics in manufacturing operations. IFAC-Papers OnLine 49:13\u201318. https:\/\/doi.org\/10.1016\/J.IFACOL.2016.12.154","journal-title":"IFAC-PapersOnLine"},{"key":"10468_CR183","doi-asserted-by":"publisher","unstructured":"Weiss BA, Alonzo D, Weinman SD (2017) Nist advanced manufacturing series 100\u201313 summary report on a workshop on advanced monitoring, diagnostics, and prognostics for manufacturing operations. National Institute of Standards and Technology, Gaithersburg. https:\/\/doi.org\/10.6028\/NIST.AMS.100-13","DOI":"10.6028\/NIST.AMS.100-13"},{"key":"10468_CR184","doi-asserted-by":"publisher","DOI":"10.1109\/TII.2021.3066257","author":"Q Wu","year":"2021","unstructured":"Wu Q, Feng Q, Ren Y et al (2021) An intelligent preventive maintenance method based on reinforcement learning for battery energy storage systems. IEEE Trans Ind Inf. https:\/\/doi.org\/10.1109\/TII.2021.3066257","journal-title":"IEEE Transactions on Industrial Informatics"},{"key":"10468_CR185","doi-asserted-by":"publisher","first-page":"576","DOI":"10.1109\/ACCESS.2017.2771827","volume":"6","author":"A Xanthopoulos","year":"2017","unstructured":"Xanthopoulos A, Kiatipis A, Koulouriotis D et al (2017) Reinforcement learning-based and parametric production-maintenance control policies for a deteriorating manufacturing system. IEEE Access 6:576\u2013588. https:\/\/doi.org\/10.1109\/ACCESS.2017.2771827","journal-title":"IEEE Access"},{"key":"10468_CR186","doi-asserted-by":"publisher","first-page":"92,110","DOI":"10.1109\/ACCESS.2019.2927426","volume":"7","author":"S Yan","year":"2019","unstructured":"Yan S, Ma B, Zheng C et al (2019) An optimal lubrication oil replacement method based on selected oil field data. IEEE Access 7:92110\u201392118. https:\/\/doi.org\/10.1109\/ACCESS.2019.2927426","journal-title":"IEEE Access"},{"key":"10468_CR187","doi-asserted-by":"publisher","unstructured":"Yang D (2022) Adaptive risk-based life-cycle management for large-scale structures using deep reinforcement learning and surrogate modeling. J Eng Mech. https:\/\/doi.org\/10.1061\/(ASCE)EM.1943-7889.0002028","DOI":"10.1061\/(ASCE)EM.1943-7889.0002028"},{"issue":"7","key":"10468_CR188","first-page":"1647","volume":"33","author":"Z Yang","year":"2013","unstructured":"Yang Z, Qi C (2013) Preventive maintenance of a multi-yield deteriorating machine: using reinforcement learning. Syst Eng Theory Pract 33(7):1647\u20131653","journal-title":"Xitong Gongcheng Lilun yu Shijian\/System Engineering Theory and Practice"},{"issue":"1","key":"10468_CR189","doi-asserted-by":"publisher","first-page":"80","DOI":"10.13196\/j.cims.2018.01.008","volume":"24","author":"H Yang","year":"2018","unstructured":"Yang H, Shen L, Cheng M et al (2018) Integrated optimization of scheduling and maintenance in multi-state production systems with deterioration effects. Comput Integr Manuf Syst (CIMS) 24(1):80\u201388. https:\/\/doi.org\/10.13196\/j.cims.2018.01.008","journal-title":"Jisuanji Jicheng Zhizao Xitong\/Computer Integrated Manufacturing Systems, CIMS"},{"key":"10468_CR190","doi-asserted-by":"publisher","unstructured":"Yang H, Li W, Wang B (2021) Joint optimization of preventive maintenance and production scheduling for multi-state production systems based on reinforcement learning. Reliab Eng Syst Saf. https:\/\/doi.org\/10.1016\/j.ress.2021.107713","DOI":"10.1016\/j.ress.2021.107713"},{"key":"10468_CR191","doi-asserted-by":"publisher","unstructured":"Zhang N, Si W (2020) Deep reinforcement learning for condition-based maintenance planning of multi-component systems under dependent competing risks. Reliab Eng Syst Saf. https:\/\/doi.org\/10.1016\/j.ress.2020.107094","DOI":"10.1016\/j.ress.2020.107094"},{"issue":"1","key":"10468_CR192","doi-asserted-by":"publisher","first-page":"156","DOI":"10.1007\/s10696-021-09403-0","volume":"34","author":"Z Zhang","year":"2022","unstructured":"Zhang Z, Tang Q (2022) Integrating preventive maintenance to two-stage assembly flow shop scheduling: Milp model, constructive heuristics and meta-heuristics. Flexible Serv Manuf J 34(1):156\u2013203. https:\/\/doi.org\/10.1007\/s10696-021-09403-0","journal-title":"Flexible Services and Manufacturing Journal"},{"key":"10468_CR193","unstructured":"Zhang C, Vinyals O, Munos R et\u00a0al (2018) A study on overfitting in deep reinforcement learning. arXiv:1804.06893"},{"key":"10468_CR194","doi-asserted-by":"publisher","unstructured":"Zhang C, Gupta C, Farahat A et\u00a0al (2019) Equipment health indicator learning using deep reinforcement learning. Lecture notes in computer science (including subseries Lecture notes in artificial intelligence and Lecture notes in bioinformatics), vol 11053. LNAI, pp 488\u2013504. https:\/\/doi.org\/10.1007\/978-3-030-10997-4_30","DOI":"10.1007\/978-3-030-10997-4_30"},{"key":"10468_CR195","doi-asserted-by":"publisher","unstructured":"Zhang P, Zhu X, Xie M (2021) A model-based reinforcement learning approach for maintenance optimization of degrading systems in a large state space. Comput Ind Eng. https:\/\/doi.org\/10.1016\/j.cie.2021.107622","DOI":"10.1016\/j.cie.2021.107622"},{"key":"10468_CR196","doi-asserted-by":"crossref","unstructured":"Zheng S, Ristovski K, Farahat A et\u00a0al (2017a) Long short-term memory network for remaining useful life estimation. In: 2017 IEEE international conference on prognostics and health management (ICPHM), pp 88\u201395","DOI":"10.1109\/ICPHM.2017.7998311"},{"key":"10468_CR197","doi-asserted-by":"crossref","unstructured":"Zheng S, Ristovski K, Farahat A et\u00a0al (2017b) Long short-term memory network for remaining useful life estimation. In: 2017 IEEE international conference on prognostics and health management (ICPHM), pp 88\u201395","DOI":"10.1109\/ICPHM.2017.7998311"},{"key":"10468_CR198","doi-asserted-by":"publisher","unstructured":"Zheng W, Lei Y, Chang Q (2017c) Reinforcement learning based real-time control policy for two-machine-one-buffer production system. In: ASME 2017 12th international manufacturing science and engineering conference, MSEC 2017 collocated with the JSME\/ASME 2017 6th international conference on materials and processing 3. https:\/\/doi.org\/10.1115\/MSEC2017-2771","DOI":"10.1115\/MSEC2017-2771"},{"key":"10468_CR199","doi-asserted-by":"publisher","unstructured":"Zonta T, da\u00a0Costa C, da\u00a0Rosa\u00a0Righi R et\u00a0al (2020) Predictive maintenance in the industry 4.0: A systematic literature review. Comput Ind Eng. https:\/\/doi.org\/10.1016\/j.cie.2020.106889","DOI":"10.1016\/j.cie.2020.106889"}],"container-title":["Artificial Intelligence Review"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-023-10468-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10462-023-10468-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-023-10468-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,17]],"date-time":"2024-10-17T00:46:18Z","timestamp":1729125978000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10462-023-10468-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,3,25]]},"references-count":199,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2023,11]]}},"alternative-id":["10468"],"URL":"https:\/\/doi.org\/10.1007\/s10462-023-10468-6","relation":{},"ISSN":["0269-2821","1573-7462"],"issn-type":[{"value":"0269-2821","type":"print"},{"value":"1573-7462","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,3,25]]},"assertion":[{"value":"9 March 2023","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 March 2023","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}