{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,15]],"date-time":"2026-05-15T17:00:04Z","timestamp":1778864404327,"version":"3.51.4"},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2025,6,9]],"date-time":"2025-06-09T00:00:00Z","timestamp":1749427200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2025,6,9]],"date-time":"2025-06-09T00:00:00Z","timestamp":1749427200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Artif Intell Rev"],"DOI":"10.1007\/s10462-025-11260-4","type":"journal-article","created":{"date-parts":[[2025,6,9]],"date-time":"2025-06-09T06:05:50Z","timestamp":1749449150000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["HIAT: human-in-the-loop reinforcement learning with auxiliary task"],"prefix":"10.1007","volume":"58","author":[{"given":"Bo","family":"Niu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Biao","family":"Luo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yongzheng","family":"Cui","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaodong","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuqian","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yu","family":"Feng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,6,9]]},"reference":[{"key":"11260_CR1","doi-asserted-by":"crossref","unstructured":"Bobu A, Wiggert M, Tomlin C, Dragan AD (2021) Feature expansive reward learning: Rethinking human input. In: Proceedings of the 2021 ACM\/IEEE international conference on human-robot interaction, pp. 216\u2013224","DOI":"10.1145\/3434073.3444667"},{"issue":"12","key":"11260_CR2","doi-asserted-by":"publisher","first-page":"5369","DOI":"10.1109\/TNNLS.2021.3084198","volume":"32","author":"Z Cao","year":"2021","unstructured":"Cao Z, Wong K, Lin C (2021) Weak human preference supervision for deep reinforcement learning. IEEE Trans Neural Netw Learn Syst 32(12):5369\u20135378","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"11260_CR3","unstructured":"Christiano PF, Leike J, Brown TB, Martic M, Legg S, Amodei D (2017) Deep reinforcement learning from human preferences. In: Neural information processing systems, pp. 4299\u20134307"},{"key":"11260_CR4","unstructured":"Conneau A, Lample G (2019) Cross-lingual language model pretraining. In: Proceedings of the 33rd international conference on neural information processing systems, pp. 7059\u20137069"},{"key":"11260_CR5","doi-asserted-by":"publisher","first-page":"221","DOI":"10.1016\/j.ins.2023.03.019","volume":"632","author":"J Deng","year":"2023","unstructured":"Deng J, Sierla S, Sun J, Vyatkin V (2023) Offline reinforcement learning for industrial process control: a case study from steel industry. Inf Sci 632:221\u2013231","journal-title":"Inf Sci"},{"key":"11260_CR6","unstructured":"Du Y, Czarnecki WM, Jayakumar SM, Farajtabar M, Pascanu R, Lakshminarayanan B (2018) Adapting auxiliary losses using gradient similarity. arXiv preprint \"http:\/\/arxiv.org\/abs\/1812.02224\""},{"key":"11260_CR7","unstructured":"Fujimoto S, Hoof H, Meger D (2018) Addressing function approximation error in actor-critic methods. In: International conference on machine learning, pp. 1582\u20131591"},{"key":"11260_CR8","unstructured":"Haarnoja T, Zhou A, Abbeel P, Levine S (2018) Soft actor-critic: off-policy maximum entropy deep reinforcement learning with a stochastic actor. In: International conference on machine learning, pp. 1861\u20131870. PMLR"},{"key":"11260_CR9","doi-asserted-by":"crossref","unstructured":"Hester T, Vecerik M, Pietquin O, Lanctot M, Schaul T, Piot B, Horgan D, Quan J, Sendonaris A, Osband I, et al (2018) Deep Q-learning from demonstrations. In: Proceedings of the AAAI conference on artificial intelligence, 32, pp 3223\u20133230","DOI":"10.1609\/aaai.v32i1.11757"},{"key":"11260_CR10","doi-asserted-by":"publisher","unstructured":"He Q, Su H, Zhang J, Hou X (2023) Frustratingly easy regularization on representation can boost deep reinforcement learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 20215\u201320225. https:\/\/doi.org\/10.48550\/arXiv.2205.14557","DOI":"10.48550\/arXiv.2205.14557"},{"issue":"10","key":"11260_CR11","doi-asserted-by":"publisher","first-page":"12098","DOI":"10.1109\/TPAMI.2023.3283537","volume":"45","author":"T Hu","year":"2023","unstructured":"Hu T, Luo B, Yang C, Huang T (2023) MO-MIX: multi-objective multi-agent cooperative decision-making with deep reinforcement learning. IEEE Trans Pattern Anal Mach Intell 45(10):12098\u201312112","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"11260_CR12","unstructured":"Kim C, Park J, Shin J, Lee H, Abbeel P, Lee K (2023) Preference transformer: modeling human preferences using transformers for RL. In: International conference on learning representations"},{"key":"11260_CR13","doi-asserted-by":"publisher","unstructured":"Lange M, Krystiniak N, Engelhardt RC, Konen W, Wiskott L (2023) Improving reinforcement learning efficiency with auxiliary tasks in non-visual environments: a comparison. arXiv e-prints, 2310 https:\/\/doi.org\/10.1007\/978-3-031-53966-4_14","DOI":"10.1007\/978-3-031-53966-4_14"},{"key":"11260_CR14","first-page":"4772","volume":"32","author":"X Lin","year":"2019","unstructured":"Lin X, Baweja H, Kantor G, Held D (2019) Adaptive auxiliary task weighting for reinforcement learning. Adv Neural Inf Process Syst 32:4772\u20134783","journal-title":"Adv Neural Inf Process Syst"},{"key":"11260_CR15","unstructured":"Liu X, Xu F, Zhang X, Liu T, Jiang S, Chen R, Zhang Z, Yu Y (2023) How to guide your learner: imitation learning with active adaptive expert involvement. In: Proceedings of the 2023 international conference on autonomous agents and multiagent systems, pp. 1276\u20131284"},{"issue":"11","key":"11260_CR16","doi-asserted-by":"publisher","first-page":"15735","DOI":"10.1109\/TNNLS.2023.3289315","volume":"35","author":"B Luo","year":"2024","unstructured":"Luo B, Wu Z, Zhou F, Wang B-C (2024) Human-in-the-loop reinforcement learning in continuous-action space. IEEE Trans Neural Netw Learn Syst 35(11):15735\u201315744","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"issue":"3","key":"11260_CR17","first-page":"510","volume":"51","author":"B Luo","year":"2025","unstructured":"Luo B, Hu T-M, Zhou Y-H, Huang T-W, Yang C-H, Gui W-H (2025) Survey on multi-agent reinforcement learning for control and decision-making. Acta Automat Sin 51(3):510\u2013539","journal-title":"Acta Automat Sin"},{"key":"11260_CR18","unstructured":"Lyle C, Rowland M, Ostrovski G, Dabney W (2021) On the effect of auxiliary tasks on representation dynamics. In: International conference on artificial intelligence and statistics. PMLR, pp. 1\u20139"},{"key":"11260_CR19","first-page":"2579","volume":"9","author":"L Maaten","year":"2008","unstructured":"Maaten L, Hinton G (2008) Visualizing data using t-SNE. J Mach Learn Res 9:2579\u20132605","journal-title":"J Mach Learn Res"},{"key":"11260_CR20","doi-asserted-by":"crossref","unstructured":"Mandel T, Liu Y, Brunskill E, Popovi\u0107 Z (2017) Where to add actions in human-in-the-loop reinforcement learning. In: Proceedings of the AAAI conference on artificial intelligence, 31","DOI":"10.1609\/aaai.v31i1.10945"},{"key":"11260_CR21","doi-asserted-by":"publisher","first-page":"358","DOI":"10.1016\/j.ins.2021.03.043","volume":"569","author":"A Perrusqu\u00faa","year":"2021","unstructured":"Perrusqu\u00faa A, Yu W, Li X (2021) Nonlinear control using human behavior learning. Inf Sci 569:358\u2013375","journal-title":"Inf Sci"},{"key":"11260_CR22","doi-asserted-by":"publisher","first-page":"331","DOI":"10.1016\/S0927-0507(05)80172-0","volume":"2","author":"ML Puterman","year":"1990","unstructured":"Puterman ML (1990) Markov decision processes. Handbooks Oper Res Manag Sci 2:331\u2013434","journal-title":"Handbooks Oper Res Manag Sci"},{"key":"11260_CR23","doi-asserted-by":"crossref","unstructured":"Rajeswaran A, Kumar V, Gupta A, Vezzani G, Schulman J, Todorov E, Levine S (2018) Learning complex dexterous manipulation with deep reinforcement learning and demonstrations. In: Proceedings of Robotics: science and systems","DOI":"10.15607\/RSS.2018.XIV.049"},{"key":"11260_CR24","unstructured":"Saunders W, Sastry G, Stuhlm\u00fcller A, Evans O (2018) Trial without error: towards safe reinforcement learning via human intervention. In: Proceedings of the 17th international conference on autonomous agents and multiagent systems, pp. 2067\u20132069"},{"key":"11260_CR25","unstructured":"Shelhamer E, Mahmoudieh P, Argus M, Darrell T (2017) Loss is its own reward: self-supervision for reinforcement learning. In: International conference on learning representations (Workshop)"},{"issue":"7587","key":"11260_CR26","doi-asserted-by":"publisher","first-page":"484","DOI":"10.1038\/nature16961","volume":"529","author":"D Silver","year":"2016","unstructured":"Silver D, Huang A, Maddison CJ, Guez A, Sifre L, Driessche G, Schrittwieser J, Antonoglou I, Panneershelvam V, Lanctot M, Dieleman S, Grewe D, Nham J, Kalchbrenner N, Sutskever I, Lillicrap TP, Leach M, Kavukcuoglu K, Graepel T, Hassabis D (2016) Mastering the game of go with deep neural networks and tree search. Nature 529(7587):484\u2013489","journal-title":"Nature"},{"key":"11260_CR27","volume-title":"Reinforcement learning: an introduction","author":"RS Sutton","year":"1998","unstructured":"Sutton RS, Barto AG (1998) Reinforcement learning: an introduction. The MIT Press, Cambridge"},{"key":"11260_CR28","unstructured":"Tangkaratt V, Han B, Khan ME, Sugiyama M (2020) Variational imitation learning with diverse-quality demonstrations. In: International conference on machine learning. PMLR, pp. 9407\u20139417"},{"key":"11260_CR29","doi-asserted-by":"crossref","unstructured":"Todorov E, Erez T, Tassa Y (2012) MuJoCo: A physics engine for model-based control. In: 2012 IEEE\/RSJ international conference on intelligent robots and systems, pp. 5026\u20135033","DOI":"10.1109\/IROS.2012.6386109"},{"key":"11260_CR30","doi-asserted-by":"crossref","unstructured":"Van\u00a0Hasselt H, Guez A, Silver D (2016) Deep reinforcement learning with double Q-learning. In: Proceedings of the AAAI conference on artificial intelligence, 30, pp 2094\u20132100","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"11260_CR31","doi-asserted-by":"crossref","unstructured":"Warnell G, Waytowich NR, Lawhern V, Stone P (2018) Deep TAMER: interactive agent shaping in high-dimensional state spaces. In: Proceedings of the AAAI conference on artificial intelligence, pp. 1545\u20131554","DOI":"10.1609\/aaai.v32i1.11485"},{"key":"11260_CR32","doi-asserted-by":"publisher","first-page":"75","DOI":"10.1016\/j.eng.2022.05.017","volume":"21","author":"J Wu","year":"2023","unstructured":"Wu J, Huang Z, Hu Z, Lv C (2023) Toward human-in-the-loop AI: enhancing deep reinforcement learning via real-time human guidance for autonomous driving. Engineering 21:75\u201391","journal-title":"Engineering"},{"key":"11260_CR33","doi-asserted-by":"crossref","unstructured":"Xin X, Karatzoglou A, Arapakis I, Jose JM (2020) Self-supervised reinforcement learning for recommender systems. In: Proceedings of the 43rd international ACM SIGIR conference on research and development in information retrieval, pp. 931\u2013940","DOI":"10.1145\/3397271.3401147"},{"key":"11260_CR34","doi-asserted-by":"crossref","unstructured":"Xu C, Tan RT, Tan Y, Chen S, Wang X, Wang Y (2023) Auxiliary tasks benefit 3d skeleton-based human motion prediction. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 9509\u20139520","DOI":"10.1109\/ICCV51070.2023.00872"},{"key":"11260_CR35","doi-asserted-by":"publisher","DOI":"10.1016\/j.ins.2024.120362","volume":"665","author":"P Yin","year":"2024","unstructured":"Yin P, Sun Y, Gao Z, Wang R, Yao Y (2024) Maint: A multi-task learning model with automatic feature interaction learning for personalized recommendations. Inf Sci 665:120362","journal-title":"Inf Sci"},{"key":"11260_CR36","doi-asserted-by":"crossref","unstructured":"Yu Y (2018) Towards sample efficient reinforcement learning. In: Proceedings of the 27th international joint conference on artificial intelligence, pp. 5739\u20135743","DOI":"10.24963\/ijcai.2018\/820"},{"key":"11260_CR37","doi-asserted-by":"publisher","DOI":"10.1016\/j.ins.2023.119684","volume":"649","author":"J Zheng","year":"2023","unstructured":"Zheng J, Jia R, Liu S, He D, Li K, Wang F (2023) Safe reinforcement learning for industrial optimal control: a case study from metallurgical industry. Inf Sci 649:119684","journal-title":"Inf Sci"},{"issue":"12","key":"11260_CR38","doi-asserted-by":"publisher","first-page":"18525","DOI":"10.1109\/TNNLS.2023.3317353","volume":"35","author":"F Zhou","year":"2024","unstructured":"Zhou F, Luo B, Wu Z, Huang T (2024) SMONAC: supervised multiobjective negative actor\u2013critic for sequential recommendation. IEEE Trans Neural Netw Learn Syst 35(12):18525\u201318537","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"issue":"9","key":"11260_CR39","doi-asserted-by":"publisher","first-page":"4261","DOI":"10.1109\/TCSI.2024.3426982","volume":"71","author":"Y Zhou","year":"2024","unstructured":"Zhou Y, Luo B, Wang X, Xu X, Xiao L (2024) RL-based adaptive optimal bipartite consensus control for nonlinear heterogeneous mass via event-triggered state feedback. IEEE Trans Circ Syst I Regul Pap 71(9):4261\u20134273","journal-title":"IEEE Trans Circ Syst I Regul Pap"},{"issue":"9","key":"11260_CR40","doi-asserted-by":"publisher","first-page":"14043","DOI":"10.1109\/TITS.2021.3134702","volume":"23","author":"Z Zhu","year":"2022","unstructured":"Zhu Z, Zhao H (2022) A survey of deep RL and IL for autonomous driving policy learning. IEEE Trans Intell Transp Syst 23(9):14043\u201314065","journal-title":"IEEE Trans Intell Transp Syst"}],"container-title":["Artificial Intelligence Review"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-025-11260-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10462-025-11260-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-025-11260-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T18:17:44Z","timestamp":1757182664000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10462-025-11260-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,9]]},"references-count":40,"journal-issue":{"issue":"9","published-online":{"date-parts":[[2025,9]]}},"alternative-id":["11260"],"URL":"https:\/\/doi.org\/10.1007\/s10462-025-11260-4","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-5823719\/v1","asserted-by":"object"}]},"ISSN":["1573-7462"],"issn-type":[{"value":"1573-7462","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,6,9]]},"assertion":[{"value":"11 May 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 June 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"273"}}