{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T21:11:52Z","timestamp":1778101912383,"version":"3.51.4"},"reference-count":35,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/100007219","name":"Natural Science Foundation of Shanghai Municipality","doi-asserted-by":"publisher","award":["23ZR1428100"],"award-info":[{"award-number":["23ZR1428100"]}],"id":[{"id":"10.13039\/100007219","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100018625","name":"Science and Technology Innovation Plan Of Shanghai Science and Technology Commission","doi-asserted-by":"publisher","award":["22511103600"],"award-info":[{"award-number":["22511103600"]}],"id":[{"id":"10.13039\/501100018625","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Engineering Applications of Artificial Intelligence"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1016\/j.engappai.2026.114788","type":"journal-article","created":{"date-parts":[[2026,4,10]],"date-time":"2026-04-10T10:16:31Z","timestamp":1775816191000},"page":"114788","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"P2","title":["Reinforcement learning joint control method for strip thickness-crown based on implicit weight contraction"],"prefix":"10.1016","volume":"176","author":[{"given":"Yue","family":"Huang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2590-0307","authenticated-orcid":false,"given":"Zhen","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Di","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ershun","family":"Pan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.engappai.2026.114788_b1","first-page":"7436","article-title":"Uncertainty-based offline reinforcement learning with diversified q-ensemble","volume":"34","author":"An","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2026.114788_b2","doi-asserted-by":"crossref","DOI":"10.1016\/j.jprocont.2025.103535","article-title":"An offline-to-online reinforcement learning framework with trajectory-guided exploration for industrial process control","volume":"154","author":"Chen","year":"2025","journal-title":"J. Process Control"},{"key":"10.1016\/j.engappai.2026.114788_b3","doi-asserted-by":"crossref","first-page":"221","DOI":"10.1016\/j.ins.2023.03.019","article-title":"Offline reinforcement learning for industrial process control: A case study from steel industry","volume":"632","author":"Deng","year":"2023","journal-title":"Inform. Sci."},{"key":"10.1016\/j.engappai.2026.114788_b4","doi-asserted-by":"crossref","DOI":"10.1016\/j.asoc.2023.110547","article-title":"Mass customization with reinforcement learning: Automatic reconfiguration of a production line","volume":"145","author":"Deng","year":"2023","journal-title":"Appl. Soft Comput."},{"key":"10.1016\/j.engappai.2026.114788_b5","first-page":"20132","article-title":"A minimalist approach to offline reinforcement learning","volume":"34","author":"Fujimoto","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2026.114788_b6","series-title":"International Conference on Machine Learning","first-page":"1587","article-title":"Addressing function approximation error in actor-critic methods","author":"Fujimoto","year":"2018"},{"key":"10.1016\/j.engappai.2026.114788_b7","series-title":"Extrinsicaly rewarded soft q imitation learning with discriminator","author":"Furuyama","year":"2024"},{"issue":"2","key":"10.1016\/j.engappai.2026.114788_b8","doi-asserted-by":"crossref","first-page":"172","DOI":"10.1109\/87.664184","article-title":"Development of an optimal crown\/shape level-2 control model for rolling mills with multiple control devices","volume":"6","author":"Guo","year":"2002","journal-title":"IEEE Trans. Control Syst. Technol."},{"key":"10.1016\/j.engappai.2026.114788_b9","doi-asserted-by":"crossref","first-page":"17734","DOI":"10.1109\/TASE.2025.3585108","article-title":"A hybrid method based on multi-agent reinforcement learning and integer programming for dynamic slab design problems in steel industry","volume":"22","author":"He","year":"2025","journal-title":"IEEE Trans. Autom. Sci. Eng."},{"key":"10.1016\/j.engappai.2026.114788_b10","series-title":"Aligniql: Policy alignment in implicit q-learning through constrained optimization","author":"He","year":"2024"},{"issue":"3","key":"10.1016\/j.engappai.2026.114788_b11","doi-asserted-by":"crossref","first-page":"1213","DOI":"10.1007\/s00170-021-07912-8","article-title":"Coordinate control of strip thickness-crown-tension based on inverse linear quadratic in tandem hot rolling mill","volume":"118","author":"Ji","year":"2022","journal-title":"Int. J. Adv. Manuf. Technol."},{"key":"10.1016\/j.engappai.2026.114788_b12","article-title":"The fixed points of off-policy TD","volume":"24","author":"Kolter","year":"2011","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2026.114788_b13","first-page":"1179","article-title":"Conservative q-learning for offline reinforcement learning","volume":"33","author":"Kumar","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2026.114788_b14","series-title":"Odice: Revealing the mystery of distribution correction estimation via orthogonal-gradient update","author":"Mao","year":"2024"},{"key":"10.1016\/j.engappai.2026.114788_b15","doi-asserted-by":"crossref","first-page":"3082","DOI":"10.1109\/TRO.2024.3400975","article-title":"Safe set-based trajectory planning for robotic manipulators","volume":"40","author":"McGovern","year":"2024","journal-title":"IEEE Trans. Robot."},{"key":"10.1016\/j.engappai.2026.114788_b16","doi-asserted-by":"crossref","first-page":"248","DOI":"10.1016\/j.jmapro.2023.08.029","article-title":"Prediction of roll wear and thermal expansion based on informer network in hot rolling process and application in the control of crown and thickness","volume":"103","author":"Meng","year":"2023","journal-title":"J. Manuf. Process."},{"key":"10.1016\/j.engappai.2026.114788_b17","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.124789","article-title":"Novel shape control system of hot-rolled strip based on machine learning fused mechanism model","volume":"255","author":"Meng","year":"2024","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.engappai.2026.114788_b18","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2024.108695","article-title":"A novel deep ensemble reinforcement learning based control method for strip flatness in cold rolling steel industry","volume":"134","author":"Peng","year":"2024","journal-title":"Eng. Appl. Artif. Intell."},{"key":"10.1016\/j.engappai.2026.114788_b19","series-title":"International Conference on Machine Learning","first-page":"28701","article-title":"Policy regularization with dataset constraint for offline reinforcement learning","author":"Ran","year":"2023"},{"issue":"1","key":"10.1016\/j.engappai.2026.114788_b20","doi-asserted-by":"crossref","first-page":"13","DOI":"10.1093\/imamat\/31.1.13","article-title":"A stable and accurate numerical method to calculate the motion of a sharp interface between fluids","volume":"31","author":"Roberts","year":"1983","journal-title":"IMA J. Appl. Math."},{"key":"10.1016\/j.engappai.2026.114788_b21","series-title":"Projected off-policy Q-learning (POP-QL) for stabilizing offline reinforcement learning","author":"Roderick","year":"2023"},{"key":"10.1016\/j.engappai.2026.114788_b22","article-title":"Reinforcement learning under model mismatch","volume":"30","author":"Roy","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2026.114788_b23","doi-asserted-by":"crossref","first-page":"20596","DOI":"10.1109\/TASE.2025.3604290","article-title":"Edge delayed deep deterministic policy gradient: Efficient continuous control for edge scenarios","volume":"22","author":"Sinigaglia","year":"2025","journal-title":"IEEE Trans. Autom. Sci. Eng."},{"key":"10.1016\/j.engappai.2026.114788_b24","doi-asserted-by":"crossref","first-page":"832","DOI":"10.1016\/j.jmapro.2022.11.075","article-title":"Control strategy of multi-stand work roll bending and shifting on the crown for UVC hot rolling mill based on MOGPR approach","volume":"85","author":"Song","year":"2023","journal-title":"J. Manuf. Process."},{"key":"10.1016\/j.engappai.2026.114788_b25","doi-asserted-by":"crossref","DOI":"10.1016\/j.conengprac.2022.105071","article-title":"Iterative learning and feedback control for the curvature and contact force of a metal strip on a roll","volume":"121","author":"Stadler","year":"2022","journal-title":"Control Eng. Pract."},{"key":"10.1016\/j.engappai.2026.114788_b26","doi-asserted-by":"crossref","DOI":"10.1016\/j.conengprac.2025.106522","article-title":"Reinforcement learning based automatic tuning of PID controllers in multivariable grinding mill circuits","volume":"165","author":"Van Niekerk","year":"2025","journal-title":"Control Eng. Pract."},{"key":"10.1016\/j.engappai.2026.114788_b27","doi-asserted-by":"crossref","first-page":"451","DOI":"10.1016\/j.jmapro.2021.07.067","article-title":"Mathematical and numerical analysis of cross-directional control for SmartCrown rolls in strip mill","volume":"69","author":"Wang","year":"2021","journal-title":"J. Manuf. Process."},{"issue":"8","key":"10.1016\/j.engappai.2026.114788_b28","doi-asserted-by":"crossref","first-page":"9928","DOI":"10.1109\/TII.2024.3390625","article-title":"Controlling aluminum strip thickness by clustered reinforcement learning with real-world dataset","volume":"20","author":"Xiao","year":"2024","journal-title":"IEEE Trans. Ind. Informatics"},{"issue":"2","key":"10.1016\/j.engappai.2026.114788_b29","doi-asserted-by":"crossref","first-page":"143","DOI":"10.3390\/biomimetics8020143","article-title":"Optimization strategy of rolling mill hydraulic roll gap control system based on improved particle swarm PID algorithm","volume":"8","author":"Yu","year":"2023","journal-title":"Biomimetics"},{"key":"10.1016\/j.engappai.2026.114788_b30","first-page":"60247","article-title":"Understanding, predicting and better resolving q-value divergence in offline-rl","volume":"36","author":"Yue","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"11","key":"10.1016\/j.engappai.2026.114788_b31","doi-asserted-by":"crossref","first-page":"7277","DOI":"10.1007\/s00170-022-09239-4","article-title":"DDPG-based continuous thickness and tension coupling control for the unsteady cold rolling process","volume":"120","author":"Zeng","year":"2022","journal-title":"Int. J. Adv. Manuf. Technol."},{"key":"10.1016\/j.engappai.2026.114788_b32","unstructured":"Zhang, C., Farhat, Z.U., Atia, G.K., Wang, Y., 2025. Model-free offline reinforcement learning with enhanced robustness. In: The Thirteenth International Conference on Learning Representations."},{"key":"10.1016\/j.engappai.2026.114788_b33","article-title":"Flatness control for cold rolling process of steel strip based on DDPG with delay compensation","author":"Zhang","year":"2025","journal-title":"Measurement"},{"key":"10.1016\/j.engappai.2026.114788_b34","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2025.110678","article-title":"A goal-conditioned offline reinforcement learning algorithm and its application to quad-rotors","volume":"152","author":"Zhong","year":"2025","journal-title":"Eng. Appl. Artif. Intell."},{"key":"10.1016\/j.engappai.2026.114788_b35","series-title":"Fat-to-thin policy optimization: Offline RL with sparse policies","author":"Zhu","year":"2025"}],"container-title":["Engineering Applications of Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0952197626010705?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0952197626010705?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T20:23:01Z","timestamp":1778098981000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0952197626010705"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":35,"alternative-id":["S0952197626010705"],"URL":"https:\/\/doi.org\/10.1016\/j.engappai.2026.114788","relation":{},"ISSN":["0952-1976"],"issn-type":[{"value":"0952-1976","type":"print"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Reinforcement learning joint control method for strip thickness-crown based on implicit weight contraction","name":"articletitle","label":"Article Title"},{"value":"Engineering Applications of Artificial Intelligence","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.engappai.2026.114788","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"114788"}}