{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,2]],"date-time":"2026-03-02T22:12:19Z","timestamp":1772489539239,"version":"3.50.1"},"reference-count":68,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"3","license":[{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Artif. Intell."],"published-print":{"date-parts":[[2026,3]]},"DOI":"10.1109\/tai.2025.3592162","type":"journal-article","created":{"date-parts":[[2025,7,24]],"date-time":"2025-07-24T17:58:32Z","timestamp":1753379912000},"page":"1223-1237","source":"Crossref","is-referenced-by-count":0,"title":["On Expressivity of Height in Neural Networks"],"prefix":"10.1109","volume":"7","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3691-5141","authenticated-orcid":false,"given":"Feng-Lei","family":"Fan","sequence":"first","affiliation":[{"name":"Frontier of Artificial Networks (FAN) Lab, Department of City University, The City University of Hong Kong, Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-6576-3298","authenticated-orcid":false,"given":"Ze-Yu","family":"Li","sequence":"additional","affiliation":[{"name":"Department of Mathematics, The Chinese University of Hong Kong, Hong Kong, Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5401-7834","authenticated-orcid":false,"given":"Huan","family":"Xiong","sequence":"additional","affiliation":[{"name":"Institute of Advanced Mathematics, Harbin Institute of Technology, Harbin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0688-202X","authenticated-orcid":false,"given":"Tieyong","family":"Zeng","sequence":"additional","affiliation":[{"name":"Department of Mathematics, The Chinese University of Hong Kong, Hong Kong, Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1038\/nature14539"},{"key":"ref2","doi-asserted-by":"crossref","DOI":"10.20944\/preprints202311.0250.v1","article-title":"Beyond deep learning\u2014space, time, and emergence","author":"Wang","year":"2023"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/s0166-4115(97)80111-2"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/d14-1179"},{"key":"ref6","first-page":"19345","article-title":"Theoretically provable spiking neural networks","volume":"35","author":"Zhang","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref7","article-title":"Connected hidden neurons (CHNNET): An artificial neural network for rapid convergence","author":"Sadat Shahir","year":"2023"},{"key":"ref8","first-page":"5669","article-title":"Neural network architecture beyond width and depth","volume":"35","author":"Zhang","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.4208\/cicp.oa-2020-0149"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2017.07.002"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1016\/0893-6080(89)90020-8"},{"key":"ref12","doi-asserted-by":"crossref","DOI":"10.21203\/rs.3.rs-92324\/v1","article-title":"Quasi-equivalence of width and depth of neural networks","author":"Fan","year":"2020"},{"key":"ref13","first-page":"6232","article-title":"The expressive power of neural networks: A view from the width","volume-title":"Proc. 31st Int. Conf. Neural Inf. Process. Syst.","author":"Lu","year":"2017"},{"key":"ref14","article-title":"Minimum width for universal approximation","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Park","year":"2020"},{"key":"ref15","first-page":"22640","article-title":"Limits to depth efficiencies of self-attention","volume-title":"Advances in Neural Information Processing Systems","volume":"33","author":"Levine","year":"2020"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.4208\/cicp.OA-2020-0149"},{"key":"ref17","first-page":"639","article-title":"Optimal approximation of continuous functions by very deep ReLU networks","volume-title":"Proc. Conf. Learn. Theory (PMLR)","author":"Yarotsky","year":"2018"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1137\/20M134695X"},{"key":"ref19","first-page":"13005","article-title":"The phase diagram of approximation rates for deep neural networks","volume":"33","author":"Yarotsky","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.4208\/jcm.2007-m2019-0239"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1162\/neco_a_01364"},{"issue":"276","key":"ref22","first-page":"1","article-title":"Deep network approximation: Achieving arbitrary accuracy with fixed number of neurons","volume":"23","author":"Zhang","year":"2022","journal-title":"J. Mach. Learn. Res."},{"key":"ref23","doi-asserted-by":"crossref","first-page":"160","DOI":"10.1016\/j.neunet.2021.04.011","article-title":"Neural network approximation: Three hidden layers are enough","volume":"141","author":"Shen","year":"2021","journal-title":"Neural Netw."},{"key":"ref24","first-page":"11932","article-title":"Elementary superexpressive activations","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Yarotsky","year":"2021"},{"key":"ref25","article-title":"Multiplicative interactions and where to find them","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Jayakumar","year":"2019"},{"key":"ref26","doi-asserted-by":"crossref","first-page":"383","DOI":"10.1016\/j.neunet.2020.01.007","article-title":"Universal approximation with quadratic deep networks","volume":"124","author":"Fan","year":"2020","journal-title":"Neural Netw."},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2023.3331380"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.1907369117"},{"key":"ref29","first-page":"2664","article-title":"Depth separations in neural networks: what is actually being separated","volume-title":"Proc. Conf. Learn. Theory (PMLR)","author":"Safran","year":"2019"},{"key":"ref30","first-page":"19433","article-title":"Neural networks with small weights and depth-separation barriers","volume":"33","author":"Vardi","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1017\/9781009025096.004"},{"key":"ref32","first-page":"4195","article-title":"Size and depth separation in approximating benign functions with neural networks","volume-title":"Proc. Conf. Learn. Theory (PMLR)","author":"Vardi","year":"2021"},{"key":"ref33","first-page":"3","article-title":"Optimization-based separations for neural networks","volume-title":"Proc. Conf. Learning Theory (PMLR)","author":"Safran","year":"2022"},{"issue":"1","key":"ref34","first-page":"5309","article-title":"Depth separation beyond radial functions","volume":"23","author":"Venturi","year":"2022","journal-title":"J. Mach. Learn. Res."},{"key":"ref35","first-page":"1249","article-title":"Width is less important than depth in ReLU neural networks","volume-title":"Proc. Conf. Learn. Theory (PMLR)","author":"Vardi","year":"2022"},{"key":"ref36","article-title":"On the number of response regions of deep feed forward networks with piece-wise linear activations","author":"Pascanu","year":"2013"},{"key":"ref37","first-page":"2924","article-title":"On the number of linear regions of deep neural networks","author":"Montufar","year":"2014","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref38","article-title":"Representation benefits of deep feedforward networks","author":"Telgarsky","year":"2015"},{"key":"ref39","article-title":"Notes on the number of linear regions of deep neural networks","volume-title":"Sampling Theory Application","author":"Mont\u00fafar","year":"2017"},{"key":"ref40","first-page":"4558","article-title":"Bounding and counting linear regions of deep neural networks","volume-title":"Proc. Int. Conf. Mach. Learn. (PMLR)","author":"Serra","year":"2018"},{"key":"ref41","article-title":"Nearly-tight bounds on linear regions of piecewise linear neural networks","author":"Hu","year":"2018"},{"key":"ref42","first-page":"10514","article-title":"On the number of linear regions of convolutional neural networks","volume-title":"Proc. Int. Conf. Mach. Learn. (PMLR)","author":"Xiong","year":"2020"},{"key":"ref43","article-title":"Is deeper better only when shallow is good?","volume":"32","author":"Malach","year":"2019","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2013.2293637"},{"key":"ref45","first-page":"2847","article-title":"On the expressive power of deep neural networks","volume-title":"Proc. Int. Conf. Mach. Learn. (PMLR)","author":"Raghu","year":"2017"},{"key":"ref46","first-page":"907","article-title":"The power of depth for feedforward neural networks","volume-title":"Proc. Conf. Learn. Theory (PMLR)","author":"Eldan","year":"2016"},{"key":"ref47","first-page":"2979","article-title":"Depth-width tradeoffs in approximating natural functions with neural networks","volume-title":"Proc. Int. Conf. Mach. Learn. (PMLR)","author":"Safran","year":"2017"},{"key":"ref48","article-title":"Depth separation beyond radial functions","author":"Venturi","year":"2021","journal-title":"J. Mach. Learn. Res."},{"key":"ref49","first-page":"1517","article-title":"Benefits of depth in neural networks","volume-title":"Proc. Conf. Learn. Theory (PMLR)","author":"Telgarsky","year":"2016"},{"key":"ref50","article-title":"Understanding deep neural networks with rectified linear units","author":"Arora","year":"2016"},{"key":"ref51","first-page":"690","article-title":"Depth separation for neural networks","volume-title":"Proc. Conf. Learn. Theory (PMLR)","author":"Daniely","year":"2017"},{"key":"ref52","article-title":"The power of deeper networks for expressing natural functions","author":"Rolnick","year":"2017"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1007\/s11633-017-1054-2"},{"key":"ref54","article-title":"Efficient deep learning of GMMs","author":"Jalali","year":"2019"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN54540.2023.10191082"},{"key":"ref56","article-title":"Connected hidden neurons (CHNnet): An artificial neural network for rapid convergence","author":"Shahir","year":"2023"},{"issue":"194","key":"ref57","first-page":"1","article-title":"On the intrinsic structures of spiking neural networks","volume":"25","author":"Zhang","year":"2024","journal-title":"J. Mach. Learn. Res."},{"key":"ref58","article-title":"Three decades of activations: A comprehensive survey of 400 activation functions for neural networks","author":"Kunc","year":"2024"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1155\/2023\/3873561"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2025.107258"},{"key":"ref61","article-title":"Why deep neural networks for function approximation?","author":"Liang","year":"2016"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.2021.3062161"},{"key":"ref63","article-title":"ADAM: A method for stochastic optimization","author":"Kingma","year":"2014"},{"key":"ref64","article-title":"Learning multiple layers of features from tiny images","author":"Krizhevsky","year":"2009"},{"issue":"7","key":"ref65","first-page":"3","article-title":"Tiny ImageNet visual recognition challenge","volume":"7","author":"Le","year":"2015","journal-title":"Comput. Sci."},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01104"},{"key":"ref68","article-title":"Deepseek-v3 technical report","author":"Liu","year":"2024"}],"container-title":["IEEE Transactions on Artificial Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/9078688\/11417361\/11095854.pdf?arnumber=11095854","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,2]],"date-time":"2026-03-02T20:59:03Z","timestamp":1772485143000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11095854\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3]]},"references-count":68,"journal-issue":{"issue":"3"},"URL":"https:\/\/doi.org\/10.1109\/tai.2025.3592162","relation":{},"ISSN":["2691-4581"],"issn-type":[{"value":"2691-4581","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3]]}}}