{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T20:12:31Z","timestamp":1783800751139,"version":"3.55.0"},"reference-count":43,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100012576","name":"Guangdong Basic and Applied Basic Research Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012576","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Engineering Applications of Artificial Intelligence"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.engappai.2026.115669","type":"journal-article","created":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T08:48:03Z","timestamp":1783673283000},"page":"115669","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"P5","title":["A safety framework for autonomous underwater vehicle navigation based on safety-constrained reinforcement learning"],"prefix":"10.1016","volume":"181","author":[{"given":"Junhao","family":"Huang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tao","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jintao","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dongye","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zijian","family":"Shen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yantao","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhenglin","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.engappai.2026.115669_bib2","series-title":"Proceedings of the 34th International Conference on Machine Learning","first-page":"22","article-title":"Constrained policy optimization","author":"Achiam","year":"2017"},{"key":"10.1016\/j.engappai.2026.115669_bib1","author":"Achiam"},{"key":"10.1016\/j.engappai.2026.115669_bib3","series-title":"Constrained Markov Decision Processes","author":"Altman","year":"2021"},{"key":"10.1016\/j.engappai.2026.115669_bib4","doi-asserted-by":"crossref","first-page":"5553","DOI":"10.1109\/TAC.2025.3550850","article-title":"Safety-critical learning of robot control with temporal logic specifications","volume":"70","author":"Cai","year":"2025","journal-title":"IEEE Trans. Automat. Control"},{"key":"10.1016\/j.engappai.2026.115669_bib5","series-title":"Proceedings of the 5th Annual Learning for Dynamics and Control Conference. Presented at the Learning for Dynamics and Control Conference","first-page":"286","article-title":"In-Distribution barrier functions: self-supervised policy filters that avoid out-of-distribution states","author":"Casta\u00f1eda","year":"2023"},{"key":"10.1016\/j.engappai.2026.115669_bib6","series-title":"Robotics: Science and Systems XVI. Presented at the Robotics: Science and Systems 2020, Robotics: Science and Systems Foundation","article-title":"Reinforcement learning for safety-critical control under model uncertainty, using control lyapunov functions and control barrier functions","author":"Choi","year":"2020"},{"key":"10.1016\/j.engappai.2026.115669_bib7","doi-asserted-by":"crossref","DOI":"10.1016\/j.oceaneng.2021.110452","article-title":"AUV position tracking and trajectory control based on fast-deployed deep reinforcement learning method","volume":"245","author":"Fang","year":"2022","journal-title":"Ocean. Eng."},{"key":"10.1016\/j.engappai.2026.115669_bib8","series-title":"Handbook of Marine Craft Hydrodynamics and Motion Control","author":"Fossen","year":"2011"},{"key":"10.1016\/j.engappai.2026.115669_bib9","series-title":"Addressing Function Approximation Error in Actor-Critic Methods","author":"Fujimoto","year":"2018"},{"key":"10.1016\/j.engappai.2026.115669_bib10","series-title":"Proceedings of the 2020 Conference on Robot Learning. Presented at the Conference on Robot Learning","first-page":"1110","article-title":"Learning to walk in the real world with minimal human effort","author":"Ha","year":"2021"},{"key":"10.1016\/j.engappai.2026.115669_bib11","series-title":"Proceedings of the 35th International Conference on Machine Learning. Presented at the International Conference on Machine Learning","first-page":"1861","article-title":"Soft actor-critic: off-policy maximum entropy deep reinforcement learning with a stochastic actor","author":"Haarnoja","year":"2018"},{"key":"10.1016\/j.engappai.2026.115669_bib12","doi-asserted-by":"crossref","DOI":"10.1016\/j.apor.2022.103326","article-title":"Deep reinforcement learning for adaptive path planning and control of an autonomous underwater vehicle","volume":"129","author":"Hadi","year":"2022","journal-title":"Appl. Ocean Res."},{"key":"10.1016\/j.engappai.2026.115669_bib14","doi-asserted-by":"crossref","first-page":"4","DOI":"10.1109\/MAES.2024.3366144","article-title":"Underwater inspection and monitoring: technologies for autonomous operations","volume":"39","author":"Ioannou","year":"2024","journal-title":"IEEE Aero. Electron. Syst. Mag."},{"key":"10.1016\/j.engappai.2026.115669_bib15","doi-asserted-by":"crossref","DOI":"10.1126\/scirobotics.adi9641","article-title":"Learning robust autonomous navigation and locomotion for wheeled-legged robots","volume":"9","author":"Lee","year":"2024","journal-title":"Sci. Robot."},{"key":"10.1016\/j.engappai.2026.115669_bib16","doi-asserted-by":"crossref","DOI":"10.1016\/j.oceaneng.2023.115361","article-title":"Fixed-time velocity-free safe formation control of AUVs with actuator saturation and unknown disturbances","volume":"285","author":"Li","year":"2023","journal-title":"Ocean. Eng."},{"key":"10.1016\/j.engappai.2026.115669_bib17","doi-asserted-by":"crossref","DOI":"10.1016\/j.oceaneng.2024.118538","article-title":"General reinforcement learning control for AUV manoeuvring in turbulent flows","volume":"309","author":"Lidtke","year":"2024","journal-title":"Ocean. Eng."},{"key":"10.1016\/j.engappai.2026.115669_bib18","article-title":"Continuous control with deep reinforcement learning","author":"Lillicrap","year":"2015","journal-title":"arXiv: Learning"},{"key":"10.1016\/j.engappai.2026.115669_bib19","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2025.127256","article-title":"Enhancing collaboration in uncertain environment: multi-agent reinforcement learning for underwater monitoring","volume":"277","author":"Luvisutto","year":"2025","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.engappai.2026.115669_bib20","doi-asserted-by":"crossref","first-page":"893","DOI":"10.1109\/TIV.2023.3282681","article-title":"Neural network model-based reinforcement learning control for AUV 3-D path following","volume":"9","author":"Ma","year":"2024","journal-title":"IEEE Trans. Intell. Veh."},{"key":"10.1016\/j.engappai.2026.115669_bib21","doi-asserted-by":"crossref","first-page":"209","DOI":"10.1109\/MNET.2024.3518780","article-title":"AUV-aided deep-sea internet of things: a machine learning perspective","volume":"39","author":"Men","year":"2025","journal-title":"IEEE Network"},{"key":"10.1016\/j.engappai.2026.115669_bib22","doi-asserted-by":"crossref","first-page":"1321","DOI":"10.1109\/TAC.2022.3152724","article-title":"Safe policies for reinforcement learning via primal-dual methods","volume":"68","author":"Paternain","year":"2023","journal-title":"IEEE Trans. Automat. Control"},{"key":"10.1016\/j.engappai.2026.115669_bib23","series-title":"Verification of a six-degree of Freedom Simulation Model for the REMUS Autonomous Underwater Vehicle (Thesis)","author":"Prestero","year":"2001"},{"key":"10.1016\/j.engappai.2026.115669_bib24","series-title":"Proximal Policy Optimization Algorithms","author":"Schulman","year":"2017"},{"key":"10.1016\/j.engappai.2026.115669_bib25","doi-asserted-by":"crossref","first-page":"4823","DOI":"10.1109\/LRA.2023.3290819","article-title":"Safe artificial potential field - novel local path planning algorithm maintaining safe distance from obstacles","volume":"8","author":"Szczepanski","year":"2023","journal-title":"IEEE Rob. Autom. Lett."},{"key":"10.1016\/j.engappai.2026.115669_bib26","first-page":"1","article-title":"Fixed-time stochastic learning from Human-UAV interaction with state-input constraints","author":"Tan","year":"2025","journal-title":"IEEE Trans. Ind. Electron."},{"key":"10.1016\/j.engappai.2026.115669_bib27","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2025.113013","article-title":"Adaptive hierarchical control of quadcopters via safe reinforcement learning from human demonstration","volume":"163","author":"Tan","year":"2026","journal-title":"Eng. Appl. Artif. Intell."},{"key":"10.1016\/j.engappai.2026.115669_bib28","doi-asserted-by":"crossref","first-page":"19568","DOI":"10.1109\/TASE.2025.3596912","article-title":"Hierarchical safe reinforcement learning control for leader-follower systems with prescribed performance","volume":"22","author":"Tan","year":"2025","journal-title":"IEEE Trans. Autom. Sci. Eng."},{"key":"10.1016\/j.engappai.2026.115669_bib29","series-title":"Proceedings of the Conference on Robot Learning. Presented at the Conference on Robot Learning","first-page":"1078","article-title":"Worst cases policy gradients","author":"Tang","year":"2020"},{"key":"10.1016\/j.engappai.2026.115669_bib30","doi-asserted-by":"crossref","first-page":"7635","DOI":"10.1109\/LRA.2021.3097073","article-title":"A predictive safety filter for learning-based racing control","volume":"6","author":"Tearle","year":"2021","journal-title":"IEEE Rob. Autom. Lett."},{"key":"10.1016\/j.engappai.2026.115669_bib13","unstructured":"Thor I. Fossen, 2002. Marine control system-guidance, navigation and control of ships, rigs and underwater vehicles. Marine Cybemetics."},{"key":"10.1016\/j.engappai.2026.115669_bib31","doi-asserted-by":"crossref","DOI":"10.1016\/j.artint.2024.104201","article-title":"Modular control architecture for safe marine navigation: reinforcement learning with predictive safety filters","volume":"336","author":"Vaaler","year":"2024","journal-title":"Artif. Intell."},{"key":"10.1016\/j.engappai.2026.115669_bib32","series-title":"Robotics Research","first-page":"3","article-title":"Reciprocal n-Body collision avoidance","author":"van den Berg","year":"2011"},{"key":"10.1016\/j.engappai.2026.115669_bib33","doi-asserted-by":"crossref","DOI":"10.1016\/j.oceaneng.2023.115828","article-title":"Safety-critical control for autonomous underwater vehicles with unknown disturbance using function approximator","volume":"287","author":"Wang","year":"2023","journal-title":"Ocean. Eng."},{"key":"10.1016\/j.engappai.2026.115669_bib34","series-title":"Model Predictive Path Integral Control Using Covariance Variable Importance Sampling","author":"Williams","year":"2015"},{"key":"10.1016\/j.engappai.2026.115669_bib35","doi-asserted-by":"crossref","DOI":"10.1016\/j.oceaneng.2022.112038","article-title":"A learning method for AUV collision avoidance through deep reinforcement learning","volume":"260","author":"Xu","year":"2022","journal-title":"Ocean. Eng."},{"key":"10.1016\/j.engappai.2026.115669_bib36","first-page":"1","article-title":"Prescribed performance optimized control of UAV with robust approximate dynamic programming under disturbance","author":"Xue","year":"2025","journal-title":"IEEE Trans. Ind. Electron."},{"key":"10.1016\/j.engappai.2026.115669_bib37","article-title":"Load-dependent structural state reconstruction-oriented and reliability-based sensor placement optimization method","author":"Yang","year":"2025","journal-title":"AIAA J."},{"key":"10.1016\/j.engappai.2026.115669_bib38","doi-asserted-by":"crossref","first-page":"15886","DOI":"10.1109\/TAES.2025.3592649","article-title":"Natural Frequency- and surface accuracy-targeted uncertain On-Orbit assembly sequence planning for large space structure with topological constraints and vibration reliability","volume":"61","author":"Yang","year":"2025","journal-title":"IEEE Trans. Aero. Electron. Syst."},{"key":"10.1016\/j.engappai.2026.115669_bib39","doi-asserted-by":"crossref","DOI":"10.1016\/j.jsv.2025.119389","article-title":"Regularization method for load reconstruction with hybrid uncertainties based on interval theory and convex model theory","volume":"619","author":"Yang","year":"2025","journal-title":"J. Sound Vib."},{"key":"10.1016\/j.engappai.2026.115669_bib40","doi-asserted-by":"crossref","first-page":"19901","DOI":"10.1007\/s11071-025-11197-x","article-title":"Uncertain attitude tracking control for QUAV based on interval LQT with states reliability constraints","volume":"113","author":"Yang","year":"2025","journal-title":"Nonlinear Dyn."},{"key":"10.1016\/j.engappai.2026.115669_bib41","first-page":"10639","article-title":"WCSAC: worst-case soft actor critic for safety-constrained reinforcement learning","volume":"35","author":"Yang","year":"2021","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"10.1016\/j.engappai.2026.115669_bib42","doi-asserted-by":"crossref","DOI":"10.1016\/j.oceaneng.2023.116540","article-title":"Tracking control of AUV via novel soft actor-critic and suboptimal demonstrations","volume":"293","author":"Zhang","year":"2024","journal-title":"Ocean. Eng."},{"key":"10.1016\/j.engappai.2026.115669_bib43","doi-asserted-by":"crossref","first-page":"276","DOI":"10.1109\/JOE.2024.3455411","article-title":"An AUV-enabled dockable platform for long-term dynamic and static monitoring of marine pastures","volume":"50","author":"Zhang","year":"2025","journal-title":"IEEE J. Ocean. Eng."}],"container-title":["Engineering Applications of Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0952197626019536?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0952197626019536?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T19:26:49Z","timestamp":1783798009000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0952197626019536"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":43,"alternative-id":["S0952197626019536"],"URL":"https:\/\/doi.org\/10.1016\/j.engappai.2026.115669","relation":{},"ISSN":["0952-1976"],"issn-type":[{"value":"0952-1976","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"A safety framework for autonomous underwater vehicle navigation based on safety-constrained reinforcement learning","name":"articletitle","label":"Article Title"},{"value":"Engineering Applications of Artificial Intelligence","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.engappai.2026.115669","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"115669"}}