{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T01:07:04Z","timestamp":1740100024134,"version":"3.37.3"},"reference-count":65,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,1,10]],"date-time":"2021-01-10T00:00:00Z","timestamp":1610236800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2021,1,10]],"date-time":"2021-01-10T00:00:00Z","timestamp":1610236800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,1,10]],"date-time":"2021-01-10T00:00:00Z","timestamp":1610236800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF","doi-asserted-by":"publisher","award":["CCF-2006738"],"award-info":[{"award-number":["CCF-2006738"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,1,10]]},"DOI":"10.1109\/icpr48806.2021.9412188","type":"proceedings-article","created":{"date-parts":[[2021,5,6]],"date-time":"2021-05-06T02:15:54Z","timestamp":1620267354000},"page":"10532-10539","source":"Crossref","is-referenced-by-count":1,"title":["RNN Training along Locally Optimal Trajectories via Frank-Wolfe Algorithm"],"prefix":"10.1109","author":[{"given":"Yun","family":"Yue","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ming","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Venkatesh","family":"Saligrama","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ziming","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","article-title":"Fastgrnn: A fast, accurate, stable and tiny kilobyte sized gated recurrent neural network","author":"kusupati","year":"2018","journal-title":"Advances in neural information processing systems"},{"key":"ref38","article-title":"Capacity and Trainability in Recurrent Neural Networks","author":"collins","year":"2016","journal-title":"ArXiv e-prints"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1989.1.2.270"},{"key":"ref32","article-title":"Stabilizing gradients for deep neural networks via efficient svd parameterization","author":"zhang","year":"2018","journal-title":"ArXiv Preprint"},{"key":"ref31","first-page":"3570","article-title":"On orthogonality and learning recurrent networks with long term dependencies","volume":"70","author":"vorontsov","year":"0","journal-title":"Proceedings of the 34th International Conference on Machine Learning-Volume"},{"key":"ref30","first-page":"4880","article-title":"Full-capacity unitary recurrent neural networks","author":"wisdom","year":"2016","journal-title":"Advances in neural information processing systems"},{"key":"ref37","first-page":"5815","article-title":"Learning long term dependencies via fourier recurrent units","author":"zhang","year":"0","journal-title":"International Conference on Machine Learning"},{"key":"ref36","first-page":"435","article-title":"Preventing gradient explosions in gated recurrent units","author":"kanai","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/W14-4012"},{"key":"ref34","first-page":"1889","article-title":"Trust region policy optimization","author":"schulman","year":"0","journal-title":"International Conference on Machine Learning"},{"key":"ref60","article-title":"Time-delay momentum: A regularization perspective on the convergence and generalization of stochastic momentum for deep learning","author":"zhang","year":"2019","journal-title":"ArXiv Preprint"},{"key":"ref62","article-title":"Sgd converges to global minimum in deep learning via star-convex path","author":"zhou","year":"0","journal-title":"International Conference on Learning Representations"},{"key":"ref61","first-page":"6673","article-title":"On the convergence rate of training recurrent neural networks","author":"allen-zhu","year":"2019","journal-title":"Advances in neural information processing systems"},{"key":"ref63","first-page":"427","article-title":"Revisiting frank-wolfe: Projection-free sparse convex optimization","author":"jaggi","year":"0","journal-title":"Proceedings of the 30th international conference on machine learning no CONF"},{"key":"ref28","article-title":"Kronecker recurrent units","author":"jose","year":"2017","journal-title":"ArXiv Preprint"},{"key":"ref64","first-page":"9017","article-title":"Fastgrnn: A fast, accurate, stable and tiny kilobyte sized gated recurrent neural network","author":"kusupati","year":"2018","journal-title":"Advances in neural information processing systems"},{"key":"ref27","first-page":"1733","article-title":"Tunable efficient unitary neural networks (eunn) and their application to rnns","author":"jing","year":"0","journal-title":"International Conference on Machine Learning"},{"key":"ref65","article-title":"Rnns incrementally evolving on an equilibrium manifold: A panacea for vanishing and exploding gradients?","author":"kag","year":"0","journal-title":"International Conference on Learning Representations"},{"key":"ref29","first-page":"2401","article-title":"Efficient orthogonal parametrisation of recurrent neural networks using householder reflections","volume":"70","author":"mhammedi","year":"0","journal-title":"Proceedings of the 34th International Conference on Machine Learning-Volume"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/21.87056"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.1983.6313075"},{"key":"ref20","article-title":"Why gradient clipping accelerates training: A theoretical justification for adaptivity","author":"zhang","year":"0","journal-title":"International Conference on Learning Representations"},{"key":"ref22","article-title":"Tutorial on training recurrent neural networks, covering BPPT, RTRL, EKF and the&#x201D; echo state network&#x201D; approach","volume":"5","author":"jaeger","year":"2002","journal-title":"GMD-Forschungszentrum Informationstechnik Bonn"},{"journal-title":"A simple algorithm for nuclear norm regularized problems","year":"2010","author":"jaggi","key":"ref21"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00572"},{"key":"ref23","article-title":"Unbiasing truncated backpropagation through time","author":"tallec","year":"2017","journal-title":"ArXiv Preprint"},{"key":"ref26","first-page":"1120","article-title":"Unitary evolution recurrent neural networks","author":"arjovsky","year":"0","journal-title":"International Conference on Machine Learning"},{"key":"ref25","article-title":"A simple way to initialize recurrent networks of rectified linear units","author":"le","year":"2015","journal-title":"ArXiv Preprint"},{"key":"ref50","first-page":"8624","article-title":"Ad-vances in optimizing recurrent networks","author":"bengio","year":"0","journal-title":"2013 IEEE International Conference on Acoustics Speech and Signal Processing"},{"key":"ref51","first-page":"77","article-title":"Dilated recurrent neural networks","author":"chang","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref59","article-title":"Adam: A method for stochastic optimization","author":"kingma","year":"2014","journal-title":"ArXiv Preprint"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/ALLERTON.2016.7852377"},{"key":"ref57","article-title":"Rnns evolving on an equilibrium manifold: A panacea for vanishing and exploding gradients?","author":"kag","year":"2019","journal-title":"ArXiv Preprint"},{"key":"ref56","article-title":"Latent odes for irregularly-sampled time series","volume":"abs 1907 3907","author":"rubanova","year":"2019","journal-title":"CoRR"},{"key":"ref55","first-page":"6571","article-title":"Neural ordinary differential equations","author":"chen","year":"2018","journal-title":"Advances in neural information processing systems"},{"key":"ref54","article-title":"Recurrent neural networks in the eye of differential equations","author":"niu","year":"2019","journal-title":"ArXiv Preprint"},{"key":"ref53","article-title":"Improving performance of recurrent neural network with relu nonlinearity","author":"talathi","year":"2015","journal-title":"ArXiv Preprint"},{"key":"ref52","article-title":"Skip rnn: Learning to skip state updates in recurrent neural networks","author":"campos","year":"2017","journal-title":"ArXiv Preprint"},{"key":"ref10","article-title":"Stable recurrent models","author":"miller","year":"2018","journal-title":"ArXiv Preprint"},{"key":"ref11","first-page":"1310","article-title":"On the difficulty of training recurrent neural networks","author":"pascanu","year":"0","journal-title":"International Conference on Machine Learning"},{"key":"ref40","article-title":"Stabilizing gradients for deep neural networks via efficient svd parameterization","author":"zhang","year":"2018","journal-title":"ICML"},{"key":"ref12","article-title":"AntisymmetricRNN: A dynamical system view on recurrent neural networks","author":"chang","year":"0","journal-title":"International Conference on Learning Representations"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1137\/0724076"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1080\/10556780410001647186"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1007\/BF01197433"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511804441"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1002\/nav.3800030109"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00348"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.81.10.3088"},{"key":"ref3","doi-asserted-by":"crossref","first-page":"2554","DOI":"10.1073\/pnas.79.8.2554","article-title":"Neural networks and physical systems with emergent collective computational abilities","volume":"79","author":"hopfield","year":"0","journal-title":"Proceedings of the National Academy of Sciences"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1016\/0893-6080(92)90011-7"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/10.52325"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ISCAS.1990.112175"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/72.105420"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2007.04.016"},{"key":"ref9","article-title":"Capacity and trainability in recurrent neural networks","author":"collins","year":"2016","journal-title":"ArXiv Preprint"},{"key":"ref46","article-title":"Quasi-recurrent neural networks","volume":"abs 1611 1576","author":"bradbury","year":"2016","journal-title":"CoRR"},{"key":"ref45","first-page":"5915","article-title":"Fast-slow recurrent neural networks","author":"mujika","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref48","article-title":"Strongly-typed recurrent neural networks","author":"balduzzi","year":"2016","journal-title":"ArXiv Preprint"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1477"},{"key":"ref42","first-page":"4785","article-title":"Resurrecting the sigmoid in deep learning through dynamical isometry: theory and practice","author":"pennington","year":"2017","journal-title":"Advances in Neural IInformation Processing Systems"},{"key":"ref41","article-title":"Efficient orthogonal parametrisation of recurrent neural networks using householder reflections","volume":"abs 1612 188","author":"mhammedi","year":"2016","journal-title":"CoRR"},{"key":"ref44","first-page":"4189","article-title":"Recurrent highway networks","author":"zilly","year":"2017","journal-title":"ICML JMLR org"},{"key":"ref43","article-title":"How to construct deep recurrent neural networks","author":"pascanu","year":"2013","journal-title":"ArXiv Preprint"}],"event":{"name":"2020 25th International Conference on Pattern Recognition (ICPR)","start":{"date-parts":[[2021,1,10]]},"location":"Milan, Italy","end":{"date-parts":[[2021,1,15]]}},"container-title":["2020 25th International Conference on Pattern Recognition (ICPR)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9411940\/9411911\/09412188.pdf?arnumber=9412188","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,5,10]],"date-time":"2022-05-10T15:40:54Z","timestamp":1652197254000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9412188\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,1,10]]},"references-count":65,"URL":"https:\/\/doi.org\/10.1109\/icpr48806.2021.9412188","relation":{},"subject":[],"published":{"date-parts":[[2021,1,10]]}}}