{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T07:46:20Z","timestamp":1740123980900,"version":"3.37.3"},"reference-count":32,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T00:00:00Z","timestamp":1727740800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T00:00:00Z","timestamp":1727740800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U1936215"],"award-info":[{"award-number":["U1936215"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Optim Theory Appl"],"published-print":{"date-parts":[[2024,10]]},"DOI":"10.1007\/s10957-024-02521-3","type":"journal-article","created":{"date-parts":[[2024,10,14]],"date-time":"2024-10-14T14:02:52Z","timestamp":1728914572000},"page":"488-528","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Linear RNNs Provably Learn Linear Dynamical Systems"],"prefix":"10.1007","volume":"203","author":[{"given":"Lifu","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tianyu","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shengwei","family":"Yi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Shen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Hu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xing","family":"Cao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,10,14]]},"reference":[{"key":"2521_CR1","unstructured":"Allen-Zhu, Z., Li, Y.: Can sgd learn recurrent neural networks with provable generalization? In: Advances in Neural Information Processing Systems, vol.\u00a032. Curran Associates Inc. (2019)"},{"key":"2521_CR2","unstructured":"Allen-Zhu, Z., Li, Y., Liang, Y.: Learning and generalization in overparameterized neural networks, going beyond two layers. In: Advances in Neural Information Processing Systems, vol.\u00a032. Curran Associates Inc. (2019)"},{"key":"2521_CR3","unstructured":"Allen-Zhu, Z., Li, Y., Song, Z.: A convergence theory for deep learning via over-parameterization. In: Chaudhuri, K., Salakhutdinov, R. (eds) Proceedings of the 36th International Conference on Machine Learning, ICML 2019, 9\u201315 June 2019, Long Beach, California, USA, volume\u00a097 of Proceedings of Machine Learning Research, pp. 242\u2013252. PMLR (2019)"},{"key":"2521_CR4","unstructured":"Allen-Zhu, Z., Li, Y., Song, Z.: On the convergence rate of training recurrent neural networks. In: Advances in Neural Information Processing Systems, vol.\u00a032. Curran Associates Inc. (2019)"},{"key":"2521_CR5","unstructured":"Belanger, D., Kakade, S.\u00a0M.: A linear dynamical system model for text. In: Bach, F.\u00a0R., Blei, D.\u00a0M. (eds) Proceedings of the 32nd International Conference on Machine Learning, ICML 2015, Lille, France, 6\u201311 July 2015, volume\u00a037 of JMLR Workshop and Conference Proceedings, pp. 833\u2013842. JMLR.org (2015)"},{"issue":"8","key":"2521_CR6","doi-asserted-by":"publisher","first-page":"1329","DOI":"10.1109\/TAC.2002.800750","volume":"47","author":"MC Campi","year":"2002","unstructured":"Campi, M.C., Weyer, E.: Finite sample properties of system identification methods. IEEE Trans. Autom. Control 47(8), 1329\u20131334 (2002)","journal-title":"IEEE Trans. Autom. Control"},{"key":"2521_CR7","unstructured":"Cao, Y., Gu, Q.: Generalization bounds of stochastic gradient descent for wide and deep neural networks. In: Advances in Neural Information Processing Systems, vol.\u00a032. Curran Associates Inc. (2019)"},{"key":"2521_CR8","doi-asserted-by":"crossref","unstructured":"Chen, T., Ljung, L.: Regularized system identification using orthonormal basis functions. In: 14th European Control Conference, ECC 2015, Linz, Austria, July 15\u201317, 2015, pp. 1291\u20131296. IEEE (2015)","DOI":"10.1109\/ECC.2015.7330716"},{"key":"2521_CR9","unstructured":"Dowler, D.\u00a0A.: Bounding the norm of matrix powers. Master\u2019s thesis, Brigham Young University (2013)"},{"key":"2521_CR10","unstructured":"Du, S.\u00a0S., Hu, W.: Width provably matters in optimization for deep linear neural networks. In: International Conference on Machine Learning (2019)"},{"key":"2521_CR11","unstructured":"Du, S.\u00a0S., Lee, J.\u00a0D., Tian, Y.:When is a convolutional filter easy to learn. In: 6th International Conference on Learning Representations, (ICLR) 2018, Vancouver, Canada, April 30\u2013May 3, 2018, Conference Track Proceedings (2018)"},{"key":"2521_CR12","unstructured":"Du, S.\u00a0S., Zhai, X., Poczos, B., Singh, A.: Gradient descent provably optimizes over-parameterized neural networks. In: 7th International Conference on Learning Representations, ICLR 2019, New Orleans, LA, USA, May 6\u20139, 2019. OpenReview.net (2019)"},{"issue":"3","key":"2521_CR13","doi-asserted-by":"publisher","first-page":"69","DOI":"10.1109\/MCS.2010.936465","volume":"30","author":"MS Grewal","year":"2010","unstructured":"Grewal, M.S., Andrews, A.P.: Applications of Kalman filtering in aerospace 1960 to the present [historical perspectives]. IEEE Control Syst. Mag. 30(3), 69\u201378 (2010)","journal-title":"IEEE Control Syst. Mag."},{"key":"2521_CR14","unstructured":"Hardt, M., Ma, T., Recht, B.: Gradient descent learns linear dynamical systems. J. Mach. Learn. Res. 19 (2016)"},{"key":"2521_CR15","unstructured":"Hazan, E., Lee, H., Singh, K., Zhang, C., Zhang, Y.: Spectral filtering for general linear dynamical systems. In: Bengio, S., Wallach, H., Larochelle, H., Grauman, K., Cesa-Bianchi, N., Garnett, R. (eds) Advances in Neural Information Processing Systems, volume\u00a031. Curran Associates Inc. (2018)"},{"key":"2521_CR16","unstructured":"Hazan, E., Singh, K., Zhang, C.: Learning linear dynamical systems via spectral filtering. In: Guyon, I., Luxburg, U.\u00a0V., Bengio, S., Wallach, H., Fergus, R., Vishwanathan, S., Garnett, R. (eds) Advances in Neural Information Processing Systems, volume\u00a030. Curran Associates Inc. (2017)"},{"key":"2521_CR17","doi-asserted-by":"publisher","unstructured":"Kalman, R.\u00a0E.: A new approach to linear filtering and prediction problems. J. Basic Eng. 82(1), 35\u201345 (1960). https:\/\/doi.org\/10.1115\/1.3662552","DOI":"10.1115\/1.3662552"},{"key":"2521_CR18","unstructured":"Kawaguchi, K.: Deep learning without poor local minima. In: Lee, D., Sugiyama, M., Luxburg, U., Guyon, I., Garnett, R. (eds) Advances in Neural Information Processing Systems, vol.\u00a029. Curran Associates Inc. (2016)"},{"key":"2521_CR19","unstructured":"Liang, P.: Statistical learning theory (winter 2016) Lecture notes for the course CS229T\/STAT231 of Stanford University (2016). https:\/\/web.stanford.edu\/class\/cs229t\/notes.pdf"},{"issue":"6","key":"2521_CR20","doi-asserted-by":"publisher","first-page":"1850","DOI":"10.1109\/TASL.2007.901312","volume":"15","author":"B Mesot","year":"2007","unstructured":"Mesot, B., Barber, D.: Switching linear dynamical systems for noise robust speech recognition. IEEE Trans. Audio Speech Lang. Process. 15(6), 1850\u20131858 (2007)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"2521_CR21","doi-asserted-by":"publisher","unstructured":"Olivier, P.: System identification using Laguerre basis functions (February 2021). https:\/\/doi.org\/10.36227\/techrxiv.13656701.v1","DOI":"10.36227\/techrxiv.13656701.v1"},{"key":"2521_CR22","doi-asserted-by":"crossref","unstructured":"Peng, B., Alcaide, E., Anthony, Q.G., Albalak, A., Arcadinho, S., Biderman, S., Cao, H., Cheng, X., Chung, M.N., Derczynski, L., Du, X., Grella, M., Gv K.\u00a0K., He, X., Hou, H., Kazienko, P., Kocon, J., Kong, J., Koptyra, B., Lau, H., Lin, J., Mantri, K.S.I., Mom, F., Saito, A., Song, G., Tang, X., Wind, J.\u00a0S., Wo\u017aniak, S., Zhang, Z., Zhou, Q., Zhu, J., Zhu, R.-J.: RWKV: Reinventing RNNs for the transformer era. In: The 2023 Conference on Empirical Methods in Natural Language Processing (2023)","DOI":"10.18653\/v1\/2023.findings-emnlp.936"},{"key":"2521_CR23","doi-asserted-by":"crossref","unstructured":"Roweis, S., Ghahramani, Z.: A unifying review of linear gaussian models. Neural Comput. (1999)","DOI":"10.1162\/089976699300016674"},{"key":"2521_CR24","unstructured":"Safran, I., Shamir, O.: Spurious local minima are common in two-layer Relu neural networks. In: International Conference on Machine Learning, pp. 4430\u20134438 (2018)"},{"key":"2521_CR25","unstructured":"Saxe, A.\u00a0M., Mcclelland, J.\u00a0L., Ganguli, S.: Exact solutions to the nonlinear dynamics of learning in deep linear neural networks. In: Bengio, Y., LeCun, Y. (eds) 2nd International Conference on Learning Representations, ICLR 2014, Banff, AB, Canada, April 14-16, 2014, Conference Track Proceedings"},{"issue":"1","key":"2521_CR26","doi-asserted-by":"publisher","first-page":"132","DOI":"10.1006\/jcss.1995.1013","volume":"50","author":"HT Siegelmann","year":"1995","unstructured":"Siegelmann, H.T., Sontag, E.D.: On the computational power of neural nets\u2014sciencedirect. J. Comput. Syst. Sci. 50(1), 132\u2013150 (1995)","journal-title":"J. Comput. Syst. Sci."},{"key":"2521_CR27","doi-asserted-by":"crossref","unstructured":"Soatto, S., Doretto, G., Wu, Y.\u00a0N.: Dynamic textures. In: Proceedings Eighth IEEE International Conference on Computer Vision. ICCV 2001, vol.\u00a02, pp. 439\u2013446 (2001)","DOI":"10.1109\/ICCV.2001.937658"},{"key":"2521_CR28","unstructured":"Tian, Y.: Symmetry-breaking convergence analysis of certain two-layered neural networks with Relu nonlinearity. In: 5th International Conference on Learning Representations, ICLR 2017, Toulon, France, April 24-26, 2017, Workshop Track Proceedings (2017)"},{"key":"2521_CR29","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.\u00a0N., Kaiser, L., Polosukhin, I.: Attention is all you need. In: Guyon, I., von Luxburg, U., Bengio, S., Wallach, H.\u00a0M., Fergus, R., Vishwanathan, S.\u00a0V.\u00a0N., Garnett, R. (eds) Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017, December 4\u20139, 2017, Long Beach, CA, USA, pp. 5998\u20136008 (2017)"},{"key":"2521_CR30","doi-asserted-by":"crossref","unstructured":"Vershynin, R.: Introduction to the non-asymptotic analysis of random matrices. In: Eldar. Y. C., Kutyniok, G. (eds) Compressed Sensing. Cambridge University Press, pp. 210\u2013268 (2012)","DOI":"10.1017\/CBO9780511794308.006"},{"key":"2521_CR31","unstructured":"Wang, L., Shen, B., Hu, B., Cao, X.: On the provable generalization of recurrent neural networks. In: Ranzato, M., Beygelzimer, A., Dauphin, Y.\u00a0N., Liang, P., Vaughan, J.\u00a0W. (eds) Advances in Neural Information Processing Systems 34: Annual Conference on Neural Information Processing Systems 2021, NeurIPS 2021, December 6\u201314, 2021, virtual, pp. 20258\u201320269 (2021)"},{"key":"2521_CR32","unstructured":"Zou, D., Long, P.M., Gu, Q.: On the global convergence of training deep linear resnets. In: 8th International Conference on Learning Representations, ICLR 2020, Addis Ababa, Ethiopia, April 26\u201330 (2020)"}],"container-title":["Journal of Optimization Theory and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10957-024-02521-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10957-024-02521-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10957-024-02521-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T18:06:57Z","timestamp":1730484417000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10957-024-02521-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10]]},"references-count":32,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2024,10]]}},"alternative-id":["2521"],"URL":"https:\/\/doi.org\/10.1007\/s10957-024-02521-3","relation":{},"ISSN":["0022-3239","1573-2878"],"issn-type":[{"type":"print","value":"0022-3239"},{"type":"electronic","value":"1573-2878"}],"subject":[],"published":{"date-parts":[[2024,10]]},"assertion":[{"value":"7 November 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 August 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 October 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}