{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,31]],"date-time":"2026-01-31T19:38:48Z","timestamp":1769888328651,"version":"3.49.0"},"reference-count":35,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"5","license":[{"start":{"date-parts":[[2024,5,1]],"date-time":"2024-05-01T00:00:00Z","timestamp":1714521600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,5,1]],"date-time":"2024-05-01T00:00:00Z","timestamp":1714521600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,5,1]],"date-time":"2024-05-01T00:00:00Z","timestamp":1714521600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62176132"],"award-info":[{"award-number":["62176132"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100020721","name":"Guoqiang Institute, Tsinghua University","doi-asserted-by":"publisher","award":["2020GQG0005"],"award-info":[{"award-number":["2020GQG0005"]}],"id":[{"id":"10.13039\/100020721","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2024,5]]},"DOI":"10.1109\/tpami.2024.3353919","type":"journal-article","created":{"date-parts":[[2024,1,15]],"date-time":"2024-01-15T21:10:11Z","timestamp":1705353011000},"page":"3972-3980","source":"Crossref","is-referenced-by-count":4,"title":["A Theoretical View of Linear Backpropagation and its Convergence"],"prefix":"10.1109","volume":"46","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5674-2252","authenticated-orcid":false,"given":"Ziang","family":"Li","sequence":"first","affiliation":[{"name":"Department of Automation, Institute for Artificial Intelligenceof Tsinghua University (THUAI), State Key Lab of Intelligent Technologies and Systems, Beijing National Research Center for Information Science and Technology (BNRist), Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0709-4877","authenticated-orcid":false,"given":"Yiwen","family":"Guo","sequence":"additional","affiliation":[{"name":"AI Lab, Bytedance Inc, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-5574-5228","authenticated-orcid":false,"given":"Haodi","family":"Liu","sequence":"additional","affiliation":[{"name":"Department of Automation, Institute for Artificial Intelligenceof Tsinghua University (THUAI), State Key Lab of Intelligent Technologies and Systems, Beijing National Research Center for Information Science and Technology (BNRist), Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8088-367X","authenticated-orcid":false,"given":"Changshui","family":"Zhang","sequence":"additional","affiliation":[{"name":"Department of Automation, Institute for Artificial Intelligenceof Tsinghua University (THUAI), State Key Lab of Intelligent Technologies and Systems, Beijing National Research Center for Information Science and Technology (BNRist), Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Backpropagating linearly improves transferability of adversarial examples","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Guo"},{"key":"ref2","article-title":"Very deep convolutional networks for large-scale image recognition","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Simonyan"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.243"},{"key":"ref5","article-title":"Attention is all you need","author":"Vaswani","year":"2017"},{"key":"ref6","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Dosovitskiy"},{"key":"ref7","article-title":"Adam: A method for stochastic optimization","author":"Kingma","year":"2014"},{"key":"ref8","article-title":"Decoupled weight decay regularization","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Loshchilov"},{"key":"ref9","article-title":"SGDR: Stochastic gradient descent with warm restarts","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Loshchilov"},{"key":"ref10","article-title":"On the convergence of Adam and beyond","author":"Reddi","year":"2019"},{"key":"ref11","first-page":"III-1139","article-title":"On the importance of initialization and momentum in deep learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Sutskever"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-7908-2604-3_16"},{"key":"ref13","first-page":"21","article-title":"A theoretical framework for back-propagation","volume-title":"Proc. Connectionist Models Summer Sch.","author":"LeCun"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/3052973.3053009"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00474"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.5244\/C.30.87"},{"key":"ref17","article-title":"Explaining and harnessing adversarial examples","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Goodfellow"},{"key":"ref18","article-title":"Adversarial machine learning at scale","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Kurakin"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.06083"},{"key":"ref20","first-page":"274","article-title":"Obfuscated gradients give a false sense of security: Circumventing defenses to adversarial examples","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Athalye"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.282"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/SP.2017.49"},{"key":"ref23","article-title":"Learning multiple layers of features from tiny images","author":"Krizhevsky","year":"2009"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"ref25","article-title":"Estimating or propagating gradients through stochastic neurons for conditional computation","author":"Bengio","year":"2013"},{"key":"ref26","first-page":"1675","article-title":"Gradient descent finds global minima of deep neural networks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Du"},{"key":"ref27","first-page":"322","article-title":"Fine-grained analysis of optimization and generalization for overparameterized two-layer neural networks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Arora"},{"key":"ref28","first-page":"3404","article-title":"An analytical formula of population gradient for two-layered ReLU network and its applications in convergence and critical point analysis","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Tian"},{"key":"ref29","article-title":"Gradient descent provably optimizes over-parameterized neural networks","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Du"},{"key":"ref30","first-page":"8024","article-title":"PyTorch: An imperative style, high-performance deep learning library","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Paszke"},{"key":"ref31","article-title":"Geometry-aware instance-reweighted adversarial training","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zhang"},{"key":"ref32","article-title":"RobustBench: A standardized adversarial robustness benchmark","author":"Croce","year":"2020"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/5.726791"},{"key":"ref34","first-page":"2206","article-title":"Reliable evaluation of adversarial robustness with an ensemble of diverse parameter-free attacks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Croce"},{"key":"ref35","article-title":"An exponential learning rate schedule for deep learning","author":"Li","year":"2019"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/34\/10490207\/10399892.pdf?arnumber=10399892","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,10]],"date-time":"2024-09-10T04:18:03Z","timestamp":1725941883000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10399892\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5]]},"references-count":35,"journal-issue":{"issue":"5"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2024.3353919","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,5]]}}}