{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T23:10:22Z","timestamp":1783984222489,"version":"3.55.0"},"reference-count":41,"publisher":"Springer Science and Business Media LLC","issue":"11","license":[{"start":{"date-parts":[[2025,10,30]],"date-time":"2025-10-30T00:00:00Z","timestamp":1761782400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,10,30]],"date-time":"2025-10-30T00:00:00Z","timestamp":1761782400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Sci. China Inf. Sci."],"published-print":{"date-parts":[[2025,11]]},"DOI":"10.1007\/s11432-025-4623-0","type":"journal-article","created":{"date-parts":[[2025,11,3]],"date-time":"2025-11-03T07:15:59Z","timestamp":1762154159000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Stackelberg games for continuous-time stochastic linear quadratic systems via Q-learning"],"prefix":"10.1007","volume":"68","author":[{"given":"Ying","family":"Cao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bing-Chang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bo","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,10,30]]},"reference":[{"key":"4623_CR1","volume-title":"Theory of Games and Economic Behavior","author":"J von Neumann","year":"1944","unstructured":"von Neumann J, Morgenstern O. Theory of Games and Economic Behavior. Princeton: Princeton University Press, 1944"},{"key":"4623_CR2","doi-asserted-by":"publisher","first-page":"55","DOI":"10.1016\/j.cobeha.2019.04.013","volume":"29","author":"K R McDonald","year":"2019","unstructured":"McDonald K R, Pearson J M. Cognitive bots and algorithmic humans: toward a shared understanding of social intelligence. Curr Opin Behav Sci, 2019, 29: 55\u201362","journal-title":"Curr Opin Behav Sci"},{"key":"4623_CR3","doi-asserted-by":"publisher","first-page":"273","DOI":"10.1007\/s10015-023-00874-y","volume":"28","author":"H Osawa","year":"2023","unstructured":"Osawa H. Human-agent interaction as augmentation of social intelligence. Artif Life Robotics, 2023, 28: 273\u2013281","journal-title":"Artif Life Robotics"},{"key":"4623_CR4","doi-asserted-by":"publisher","first-page":"33","DOI":"10.1038\/d41586-021-01170-0","volume":"593","author":"A Dafoe","year":"2021","unstructured":"Dafoe A, Bachrach Y, Hadfield G, et al. Cooperative AI: machines must learn to find common ground. Nature, 2021, 593: 33\u201336","journal-title":"Nature"},{"key":"4623_CR5","doi-asserted-by":"publisher","first-page":"1892","DOI":"10.1360\/SSI-2023-0010","volume":"53","author":"J Hao","year":"2023","unstructured":"Hao J, Shao K, Li K, et al. Research and applications of game intelligence (in Chinese). Sci Sin Inform, 2023, 53: 1892\u20131923","journal-title":"Sci Sin Inform"},{"key":"4623_CR6","doi-asserted-by":"publisher","DOI":"10.1007\/978-93-86279-17-0","volume-title":"Introduction to Game Theory","author":"S Tijs","year":"2003","unstructured":"Tijs S. Introduction to Game Theory. Gurgaon: Hindustan Book Agency, 2003"},{"key":"4623_CR7","volume-title":"The Theory of the Market Economy","author":"H V Stackelberg","year":"1952","unstructured":"Stackelberg H V. The Theory of the Market Economy. London: Oxford University Press, 1952"},{"key":"4623_CR8","doi-asserted-by":"publisher","first-page":"322","DOI":"10.1109\/TAC.1973.1100307","volume":"18","author":"M Simaan","year":"1973","unstructured":"Simaan M, Cruz J. A Stackelberg solution for games with many players. IEEE Trans Automat Contr, 1973, 18: 322\u2013324","journal-title":"IEEE Trans Automat Contr"},{"key":"4623_CR9","doi-asserted-by":"publisher","first-page":"166","DOI":"10.1109\/TAC.1979.1101999","volume":"24","author":"T Basar","year":"1979","unstructured":"Basar T, Selbuz H. Closed-loop Stackelberg strategies with applications in the optimal control of multilevel systems. IEEE Trans Automat Contr, 1979, 24: 166\u2013179","journal-title":"IEEE Trans Automat Contr"},{"key":"4623_CR10","doi-asserted-by":"publisher","first-page":"305","DOI":"10.1016\/S1474-6670(17)64812-2","volume":"13","author":"B Tolwinski","year":"1980","unstructured":"Tolwinski B. Information and dominant player solutions in linear-quadratic dynamic games. IFAC Proc Vol, 1980, 13: 305\u2013313","journal-title":"IFAC Proc Vol"},{"key":"4623_CR11","doi-asserted-by":"publisher","first-page":"621","DOI":"10.1109\/TAC.2008.917649","volume":"53","author":"M Jungers","year":"2008","unstructured":"Jungers M. On linear-quadratic Stackelberg games with time preference rates. IEEE Trans Automat Contr, 2008, 53: 621\u2013625","journal-title":"IEEE Trans Automat Contr"},{"key":"4623_CR12","doi-asserted-by":"publisher","first-page":"387","DOI":"10.1007\/s10883-011-9123-2","volume":"17","author":"M Jungers","year":"2011","unstructured":"Jungers M, Trelat E, Abou-kandil H. Min-max and min-min Stackelberg strategies with closed-loop information structure. J Dyn Control Syst, 2011, 17: 387\u2013425","journal-title":"J Dyn Control Syst"},{"key":"4623_CR13","doi-asserted-by":"publisher","first-page":"1309","DOI":"10.1109\/TAC.2023.3300365","volume":"69","author":"P Zhang","year":"2024","unstructured":"Zhang P, Zhang Y. Two-step Stackelberg approach for the two weak pursuers and one strong evader closed-loop game. IEEE Trans Automat Contr, 2024, 69: 1309\u20131315","journal-title":"IEEE Trans Automat Contr"},{"key":"4623_CR14","doi-asserted-by":"publisher","first-page":"443","DOI":"10.1007\/BF00934911","volume":"35","author":"A Bagchi","year":"1981","unstructured":"Bagchi A, Basar T. Stackelberg strategies in linear-quadratic stochastic differential games. J Optim Theor Appl, 1981, 35: 443\u2013464","journal-title":"J Optim Theor Appl"},{"key":"4623_CR15","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1155\/2019\/1798585","volume":"2019","author":"K Du","year":"2019","unstructured":"Du K, Wu Z. Linear-quadratic Stackelberg game for mean-field backward stochastic differential system and application. Math Probl Eng, 2019, 2019: 1\u201317","journal-title":"Math Probl Eng"},{"key":"4623_CR16","doi-asserted-by":"publisher","first-page":"2445","DOI":"10.1007\/s00245-020-09713-z","volume":"84","author":"J Huang","year":"2021","unstructured":"Huang J, Si K, Wu Z. Linear-quadratic mixed Stackelberg-Nash stochastic differential game with major-minor agents. Appl Math Optim, 2021, 84: 2445\u20132494","journal-title":"Appl Math Optim"},{"key":"4623_CR17","first-page":"126819","volume":"420","author":"Y Y Zheng","year":"2022","unstructured":"Zheng Y Y, Shi J T. A linear-quadratic partially observed Stackelberg stochastic differential game with application. Appl Math Comput, 2022, 420: 126819","journal-title":"Appl Math Comput"},{"key":"4623_CR18","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1007\/s00245-024-10132-7","volume":"89","author":"N Li","year":"2024","unstructured":"Li N, Wang S. Linear-quadratic stochastic Stackelberg games of N players for time-delay systems and related FBSDEs. Appl Math Optim, 2024, 89: 67","journal-title":"Appl Math Optim"},{"key":"4623_CR19","doi-asserted-by":"publisher","first-page":"1956","DOI":"10.1137\/140958906","volume":"53","author":"A Bensoussan","year":"2015","unstructured":"Bensoussan A, Chen S, Sethi S P. The maximum principle for global solutions of stochastic Stackelberg differential games. SIAM J Control Optim, 2015, 53: 1956\u20131981","journal-title":"SIAM J Control Optim"},{"key":"4623_CR20","doi-asserted-by":"publisher","first-page":"301","DOI":"10.1016\/j.automatica.2016.10.016","volume":"76","author":"H Mukaidani","year":"2017","unstructured":"Mukaidani H, Xu H. Infinite horizon linear-quadratic Stackelberg games for discrete-time stochastic systems. Automatica, 2017, 76: 301\u2013308","journal-title":"Automatica"},{"key":"4623_CR21","doi-asserted-by":"publisher","first-page":"200","DOI":"10.1016\/j.automatica.2018.08.008","volume":"97","author":"J Moon","year":"2018","unstructured":"Moon J, Basar T. Linear quadratic mean field Stackelberg differential games. Automatica, 2018, 97: 200\u2013213","journal-title":"Automatica"},{"key":"4623_CR22","doi-asserted-by":"publisher","first-page":"109406","DOI":"10.1016\/j.automatica.2020.109406","volume":"125","author":"Z Li","year":"2021","unstructured":"Li Z, Marelli D, Fu M, et al. Linear quadratic Gaussian Stackelberg game under asymmetric information patterns. Automatica, 2021, 125: 109406","journal-title":"Automatica"},{"key":"4623_CR23","volume-title":"Reinforcement Learning: An Introduction","author":"R S Sutton","year":"2018","unstructured":"Sutton R S, Barto A G. Reinforcement Learning: An Introduction. 2nd ed. Cambridge: MIT Press, 2018","edition":"2nd ed."},{"key":"4623_CR24","doi-asserted-by":"publisher","first-page":"39","DOI":"10.1109\/MCI.2009.932261","volume":"4","author":"F Y Wang","year":"2009","unstructured":"Wang F Y, Zhang H, Liu D. Adaptive dynamic programming: an introduction. IEEE Comput Intell Mag, 2009, 4: 39\u201347","journal-title":"IEEE Comput Intell Mag"},{"key":"4623_CR25","first-page":"493","volume-title":"Handbook of Intelligent Control","author":"P J Werbos","year":"1992","unstructured":"Werbos P J. Approximate dynamic programming for real-time control and neural modeling. In: Handbook of Intelligent Control. New York: Van Nostrand Reinhold, 1992. 493\u2013525"},{"key":"4623_CR26","doi-asserted-by":"publisher","first-page":"240","DOI":"10.1109\/TSMCB.2006.880135","volume":"37","author":"A Al-Tamimi","year":"2007","unstructured":"Al-Tamimi A, Abu-Khalaf M, Lewis F L. Adaptive critic designs for discrete-time zero-sum games with application to control. IEEE Trans Syst Man Cybern B, 2007, 37: 240\u2013247","journal-title":"IEEE Trans Syst Man Cybern B"},{"key":"4623_CR27","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1109\/MCAS.2009.933854","volume":"9","author":"F L Lewis","year":"2009","unstructured":"Lewis F L, Vrabie D. Reinforcement learning and adaptive dynamic programming for feedback control. IEEE Circuits Syst Mag, 2009, 9: 32\u201350","journal-title":"IEEE Circuits Syst Mag"},{"key":"4623_CR28","doi-asserted-by":"publisher","first-page":"477","DOI":"10.1016\/j.automatica.2008.08.017","volume":"45","author":"D Vrabie","year":"2009","unstructured":"Vrabie D, Pastravanu O, Abu-Khalaf M, et al. Adaptive optimal control for continuous-time linear systems based on policy iteration. Automatica, 2009, 45: 477\u2013484","journal-title":"Automatica"},{"key":"4623_CR29","doi-asserted-by":"publisher","first-page":"1176","DOI":"10.1109\/TASE.2013.2280974","volume":"11","author":"Q Wei","year":"2014","unstructured":"Wei Q, Liu D. A novel iterative \u03b8-adaptive dynamic programming for discrete-time nonlinear systems. IEEE Trans Automat Sci Eng, 2014, 11: 1176\u20131190","journal-title":"IEEE Trans Automat Sci Eng"},{"key":"4623_CR30","first-page":"124568","volume":"363","author":"X K Liu","year":"2019","unstructured":"Liu X K, Ge Y Y, Li Y. Stackelberg games for model-free continuous-time stochastic systems based on adaptive dynamic programming. Appl Math Comput, 2019, 363: 124568","journal-title":"Appl Math Comput"},{"key":"4623_CR31","doi-asserted-by":"publisher","first-page":"273","DOI":"10.1109\/TCYB.2023.3251653","volume":"54","author":"M Lin","year":"2024","unstructured":"Lin M, Zhao B, Liu D. Event-triggered robust adaptive dynamic programming for multiplayer Stackelberg-Nash games of uncertain nonlinear systems. IEEE Trans Cybern, 2024, 54: 273\u2013286","journal-title":"IEEE Trans Cybern"},{"key":"4623_CR32","first-page":"279","volume":"8","author":"C Watkins","year":"1992","unstructured":"Watkins C, Christopher J, Dayan P. Q-learning. Mach Learn, 1992, 8: 279\u2013292","journal-title":"Mach Learn"},{"key":"4623_CR33","doi-asserted-by":"publisher","first-page":"274","DOI":"10.1016\/j.automatica.2015.08.017","volume":"61","author":"K G Vamvoudakis","year":"2015","unstructured":"Vamvoudakis K G. Non-zero sum Nash Q-learning for unknown deterministic continuous-time linear systems. Automatica, 2015, 61: 274\u2013281","journal-title":"Automatica"},{"key":"4623_CR34","doi-asserted-by":"publisher","first-page":"1907","DOI":"10.1007\/s11424-024-3343-5","volume":"37","author":"B Q Zhang","year":"2024","unstructured":"Zhang B Q, Wang B C, Cao Y. An online Q-learning method for linear-quadratic nonzero-sum stochastic differential games with completely unknown dynamics. J Syst Sci Complex, 2024, 37: 1907\u20131922","journal-title":"J Syst Sci Complex"},{"key":"4623_CR35","doi-asserted-by":"publisher","first-page":"1083","DOI":"10.1360\/SSI-2021-0016","volume":"52","author":"M Li","year":"2022","unstructured":"Li M, Qin J H, Wang L. Seeking equilibrium for linear-quadratic two-player Stackelberg game: a Q-learning approach (in Chinese). Sci Sin Inform, 2022, 52: 1083\u20131097","journal-title":"Sci Sin Inform"},{"key":"4623_CR36","doi-asserted-by":"publisher","first-page":"372","DOI":"10.1137\/0309027","volume":"9","author":"R G Agniel","year":"1971","unstructured":"Agniel R G, Jury E I. Almost sure boundedness of randomly sampled systems. SIAM J Control, 1971, 9: 372\u2013384","journal-title":"SIAM J Control"},{"key":"4623_CR37","doi-asserted-by":"publisher","first-page":"87","DOI":"10.1016\/j.automatica.2003.07.002","volume":"40","author":"W Zhang","year":"2004","unstructured":"Zhang W, Chen B S. On stabilizability and exact observability of stochastic systems with their applications. Automatica, 2004, 40: 87\u201394","journal-title":"Automatica"},{"key":"4623_CR38","volume-title":"Stochastic Stability of Differential Equations","author":"R Z Khasminiskii","year":"1980","unstructured":"Khasminiskii R Z. Stochastic Stability of Differential Equations. Berlin: Springer-Verlag, 1980"},{"key":"4623_CR39","doi-asserted-by":"publisher","first-page":"779","DOI":"10.1016\/j.automatica.2004.11.034","volume":"41","author":"M Abu-Khalaf","year":"2005","unstructured":"Abu-Khalaf M, Lewis F L. Nearly optimal control laws for nonlinear systems with saturating actuators using a neural network HJB approach. Automatica, 2005, 41: 779\u2013791","journal-title":"Automatica"},{"key":"4623_CR40","doi-asserted-by":"publisher","first-page":"19","DOI":"10.1109\/37.126844","volume":"12","author":"R S Sutton","year":"1992","unstructured":"Sutton R S, Barto A G, Williams R J. Reinforcement learning is direct adaptive optimal control. IEEE Control Syst Mag, 1992, 12: 19\u201322","journal-title":"IEEE Control Syst Mag"},{"key":"4623_CR41","doi-asserted-by":"publisher","first-page":"878","DOI":"10.1016\/j.automatica.2010.02.018","volume":"46","author":"K G Vamvoudakis","year":"2010","unstructured":"Vamvoudakis K G, Lewis F L. Online actor-critic algorithm to solve the continuous-time infinite horizon optimal control problem. Automatica, 2010, 46: 878\u2013888","journal-title":"Automatica"}],"container-title":["Science China Information Sciences"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-025-4623-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11432-025-4623-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-025-4623-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,3]],"date-time":"2025-11-03T08:03:33Z","timestamp":1762157013000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11432-025-4623-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,30]]},"references-count":41,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2025,11]]}},"alternative-id":["4623"],"URL":"https:\/\/doi.org\/10.1007\/s11432-025-4623-0","relation":{},"ISSN":["1674-733X","1869-1919"],"issn-type":[{"value":"1674-733X","type":"print"},{"value":"1869-1919","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,10,30]]},"assertion":[{"value":"28 February 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 July 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 September 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 October 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"210204"}}