{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T23:10:16Z","timestamp":1772838616303,"version":"3.50.1"},"reference-count":66,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"10","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2025,10,1]]},"DOI":"10.1587\/transinf.2024edp7245","type":"journal-article","created":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T18:15:02Z","timestamp":1743099302000},"page":"1194-1205","source":"Crossref","is-referenced-by-count":1,"title":["Ultra-Fast NAS Based on Normalized Generalization Error with Random NTK"],"prefix":"10.1587","volume":"E108.D","author":[{"given":"Keigo","family":"WAKAYAMA","sequence":"first","affiliation":[{"name":"Department of Mathematical and Computing Science, Institute of Science Tokyo"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Takafumi","family":"KANAMORI","sequence":"additional","affiliation":[{"name":"Department of Mathematical and Computing Science, Institute of Science Tokyo"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"key":"1","doi-asserted-by":"publisher","unstructured":"[1] S. Pouyanfar, S. Sadiq, Y. Yan, H. Tian, Y. Tao, M.P. Reyes, M.L. Shyu, S.C. Chen, and S.S. Iyengar, \u201cA survey on deep learning: Algorithms, techniques, and applications,\u201d ACM Comput. Surv., vol.51, no.5, Sept. 2018. 10.1145\/3234150","DOI":"10.1145\/3234150"},{"key":"2","doi-asserted-by":"crossref","unstructured":"[2] K. He, X. Zhang, S. Ren, and J. Sun, \u201cDeep residual learning for image recognition,\u201d Proc. CVPR, pp.770-778, 2016. 10.1109\/CVPR.2016.90","DOI":"10.1109\/CVPR.2016.90"},{"key":"3","doi-asserted-by":"crossref","unstructured":"[3] S. Xie, R. Girshick, P. Dollar, Z. Tu, and K. He, \u201cAggregated residual transformations for deep neural networks,\u201d Proc. CVPR, pp.1492-1500, 2017. 10.1109\/CVPR.2017.634","DOI":"10.1109\/CVPR.2017.634"},{"key":"4","doi-asserted-by":"crossref","unstructured":"[4] G. Huang, Z. Liu, L. van der Maaten, and K.Q. Weinberger, \u201cDensely connected convolutional networks,\u201d Proc. CVPR, pp.4700-4708, 2017. 10.1109\/CVPR.2017.243","DOI":"10.1109\/CVPR.2017.243"},{"key":"5","unstructured":"[5] Y. Chen, J. Li, H. Xiao, X. Jin, S. Yan, and J. Feng, \u201cDual path networks,\u201d Proc. NeurIPS, pp.1-9, 2017."},{"key":"6","unstructured":"[6] Z. Allen-Zhu, Y. Li, and Z. Song, \u201cA convergence theory for deep learning via over-parameterization,\u201d Proc. ICML, pp.242-252, 2019."},{"key":"7","unstructured":"[7] Y. Lu, C. Ma, Y. Lu, J. Lu, and L. Ying, \u201cA mean field analysis of deep ResNet and beyond: Towards provably optimization via overparameterization from depth,\u201d Proc. ICML, pp.6426-6436, 2020."},{"key":"8","unstructured":"[8] A. Jacot, F. Gabriel, and C. Hongler, \u201cNeural tangent kernel: Convergence and generalization in neural networks,\u201d Proc. NeurIPS, pp.1-10, 2018."},{"key":"9","doi-asserted-by":"crossref","unstructured":"[9] J. Lee, L. Xiao, S. Schoenholz, Y. Bahri, R. Novak, J. Sohl-Dickstein, and J. Pennington, \u201cWide neural networks of any depth evolve as linear models under gradient descent,\u201d Proc. NeurIPS, pp.1-12, 2019.","DOI":"10.1088\/1742-5468\/abc62b"},{"key":"10","unstructured":"[10] K. Huang, Y. Wang, M. Tao, and T. Zhao, \u201cWhy do deep residual networks generalize better than deep feedforward networks? \u2014 A neural tangent kernel perspective,\u201d Proc. NeurIPS, pp.2698-2709, 2020."},{"key":"11","unstructured":"[11] S. Arora, S.S. Du, W. Hu, Z. Li, R.R. Salakhutdinov, and R. Wang, \u201cOn exact computation with an infinitely wide neural net,\u201d Proc. NeurIPS, pp.1-10, 2019."},{"key":"12","unstructured":"[12] S. Alemohammad, Z. Wang, R. Balestriero, and R. Baraniuk, \u201cThe recurrent neural tangent kernel,\u201d Proc. ICLR, pp.1-26, 2021."},{"key":"13","unstructured":"[13] G. Yang, \u201cTensor programs II: Neural tangent kernel for any architecture,\u201d arXiv:2006.14548, pp.1-60, 2020."},{"key":"14","unstructured":"[14] G. Yang and E. Littwin, \u201cTensor programs IIb: Architectural universality of neural tangent kernel training dynamics,\u201d Proc. ICML, pp.11762-11772, 2021."},{"key":"15","unstructured":"[15] B. Hanin and M. Nica, \u201cFinite depth and width corrections to the neural tangent kernel,\u201d Proc. ICLR, pp.1-25, 2020."},{"key":"16","unstructured":"[16] E. Littwin, T. Galanti, and L. Wolf, \u201cOn random kernels of residual architectures,\u201d Proc. UAI, pp.897-907, 2021."},{"key":"17","unstructured":"[17] Z. Hu and H. Huang, \u201cOn the random conjugate kernel and neural tangent kernel,\u201d Proc. ICML, pp.4359-4368, 2021."},{"key":"18","unstructured":"[18] M. Li, M. Nica, and D. Roy, \u201cThe future is log-Gaussian: ResNets and their infinite-depth-and-width limit at initialization,\u201d Proc. NeurIPS, pp.7852-7864, 2021."},{"key":"19","unstructured":"[19] M. Seleznova and G. Kutyniok, \u201cNeural tangent kernel beyond the infinite-width limit: Effects of depth and initialization,\u201d Proc. ICML, pp.19522-19560, 2022."},{"key":"20","unstructured":"[20] E. Littwin, B. Myara, S. Sabah, J. Susskind, S. Zhai, and O. Golan, \u201cCollegial ensembles,\u201d Proc. NeurIPS, pp.18738-18748, 2020."},{"key":"21","unstructured":"[21] J. Huang and H.-T. Yau, \u201cDynamics of deep neural networks and neural tangent hierarchy,\u201d Proc. ICML, pp.4542-4551, 2020."},{"key":"22","doi-asserted-by":"publisher","unstructured":"[22] M. Geiger, A. Jacot, S. Spigler, F. Gabriel, L. Sagun, S. d\u2019Ascoli, G. Biroli, C. Hongler, and M. Wyart, \u201cScaling description of generalization with number of parameters in deep learning,\u201d Journal of Statistical Mechanics, vol.2020, no.2, pp.1-23, Feb. 2020. 10.1088\/1742-5468\/ab633c","DOI":"10.1088\/1742-5468\/ab633c"},{"key":"23","doi-asserted-by":"crossref","unstructured":"[23] T. Elsken, J.H. Metzen, and F. Hutter, \u201cNeural architecture search: A survey,\u201d J. Mach. Learn. Res., vol.20, no.55, pp.1-21, 2019.","DOI":"10.1007\/978-3-030-05318-5_11"},{"key":"24","unstructured":"[24] H. Liu, K. Simonyan, and Y. Yang, \u201cDARTS: Differentiable architecture search,\u201d Proc. ICLR, pp.1-13, 2019."},{"key":"25","unstructured":"[25] M.S. Abdelfattah, A. Mehrotra, \u0141. Dudziak, and N.D. Lane, \u201cZero-cost proxies for lightweight NAS,\u201d Proc. ICLR, pp.1-17, 2021."},{"key":"26","unstructured":"[26] Y. Shu, S. Cai, Z. Dai, B.C. Ooi, and B.K.H. Low, \u201cNASI: Label- and data-agnostic neural architecture search at initialization,\u201d Proc. ICLR, pp.1-25, 2022."},{"key":"27","unstructured":"[27] A. Zela, J. Siems, and F. Hutter, \u201cNAS-Bench-1Shot1: Benchmarking and dissecting one-shot neural architecture search,\u201d Proc. ICLR, pp.1-20, 2020."},{"key":"28","unstructured":"[28] X. Dong and Y. Yang, \u201cNAS-Bench-201: Extending the scope of reproducible neural architecture search,\u201d Proc. ICLR, pp.1-16, 2020."},{"key":"29","unstructured":"[29] A. Vaswani, N. Shazeer, N. Parmar, J. Uszkoreit, L. Jones, A.N. Gomez, \u0141. Kaiser, and I. Polosukhin, \u201cAttention is all you need,\u201d Proc. NeurIPS, pp.1-11, 2017."},{"key":"30","doi-asserted-by":"crossref","unstructured":"[30] Z. Liu, Y. Lin, Y. Cao, H. Hu, Y. Wei, Z. Zhang, S. Lin, and B. Guo, \u201cSwin transformer: Hierarchical vision transformer using shifted windows,\u201d Proc. ICCV, pp.10012-10022, 2021. 10.1109\/ICCV48922.2021.00986","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"31","doi-asserted-by":"crossref","unstructured":"[31] K. Wakayama and S. Saito, \u201cCNN-transformer with self-attention network for sound event detection,\u201d Proc. ICASSP, pp.806-810, 2022. 10.1109\/ICASSP43922.2022.9747762","DOI":"10.1109\/ICASSP43922.2022.9747762"},{"key":"32","doi-asserted-by":"publisher","unstructured":"[32] P. Ren, Y. Xiao, X. Chang, P.-y. Huang, Z. Li, X. Chen, and X. Wang, \u201cA comprehensive survey of neural architecture search: Challenges and solutions,\u201d ACM Comput. Surv., vol.54, no.4, May 2021. 10.1145\/3447582","DOI":"10.1145\/3447582"},{"key":"33","unstructured":"[33] H. Pham, M. Guan, B. Zoph, Q. Le, and J. Dean, \u201cEfficient neural architecture search via parameters sharing,\u201d Proc. ICML, pp.4095-4104, 2018."},{"key":"34","doi-asserted-by":"crossref","unstructured":"[34] X. Dong and Y. Yang, \u201cSearching for a robust neural architecture in four GPU hours,\u201d Proc. CVPR, pp.1761-1770, 2019. 10.1109\/CVPR.2019.00186","DOI":"10.1109\/CVPR.2019.00186"},{"key":"35","unstructured":"[35] S. Xie, H. Zheng, C. Liu, and L. Lin, \u201cSNAS: Stochastic neural architecture search,\u201d Proc. ICLR, pp.1-17, 2019."},{"key":"36","unstructured":"[36] H. Wang, C. Ge, H. Chen, and X. Sun, \u201cPreNAS: Preferred one-shot learning towards efficient neural architecture search,\u201d Proc. ICML, pp.35642-35654, 2023."},{"key":"37","unstructured":"[37] Y. Shu, W. Wang, and S. Cai, \u201cUnderstanding architectures learnt by cell-based neural architecture search,\u201d Proc. ICLR, pp.1-21, 2020."},{"key":"38","unstructured":"[38] X. Wan, B. Ru, P.M. Esperan\u00e7a, and Z. Li, \u201cOn redundancy and diversity in cell-based neural architecture search,\u201d Proc. ICLR, pp.1-25, 2022."},{"key":"39","unstructured":"[39] C. Ying, A. Klein, E. Christiansen, E. Real, K. Murphy, and F. Hutter, \u201cNAS-Bench-101: Towards reproducible neural architecture search,\u201d Proc. ICML, pp.7105-7114, 2019."},{"key":"40","doi-asserted-by":"crossref","unstructured":"[40] P. Ye, B. Li, Y. Li, T. Chen, J. Fan, and W. Ouyang, \u201c<i>\u03b2<\/i>-DARTS: Beta-decay regularization for differentiable architecture search,\u201d Proc. CVPR, pp.10874-10883, 2022. 10.1109\/CVPR52688.2022.01060","DOI":"10.1109\/CVPR52688.2022.01060"},{"key":"41","doi-asserted-by":"publisher","unstructured":"[41] K. Sakamoto, H. Ishibashi, R. Sato, S. Shirakawa, Y. Akimoto, and H. Hino, \u201cATNAS: Automatic termination for neural architecture search,\u201d Neural Networks, vol.166, pp.446-458, 2023. 10.1016\/j.neunet.2023.07.011","DOI":"10.1016\/j.neunet.2023.07.011"},{"key":"42","unstructured":"[42] D.S. Park, J. Lee, D. Peng, Y. Cao, and J. Sohl-Dickstein, \u201cTowards NNGP-guided neural architecture Search,\u201d arXiv:2011.06006, pp.1-19, 2020."},{"key":"43","unstructured":"[43] J. Mellor, J. Turner, A. Storkey, and E.J. Crowley, \u201cNeural architecture search without training,\u201d Proc. ICML, pp.7588-7598, 2021."},{"key":"44","unstructured":"[44] W. Chen, X. Gong, and Z. Wang, \u201cNeural architecture search on ImageNet in four GPU hours: A theoretically inspired perspective,\u201d Proc. ICLR, pp.1-15, 2021."},{"key":"45","unstructured":"[45] J. Xu, L. Zhao, J. Lin, R. Gao, X. Sun, and H. Yang, \u201cKNAS: Green neural architecture search,\u201d Proc. ICML, pp.11613-11625, 2021."},{"key":"46","unstructured":"[46] Z. Zhang and Z. Jia, \u201cGradSign: Model performance inference with theoretical insights,\u201d Proc. ICLR, pp.1-21, 2022."},{"key":"47","unstructured":"[47] G. Li, Y. Yang, K. Bhardwaj, and R. Marculescu, \u201cZiCo: Zero-shot NAS via inverse coefficient of variation on gradients,\u201d Proc. ICLR, pp.1-31, 2023."},{"key":"48","unstructured":"[48] Y. Peng, A. Song, V. Ciesielski, H.M. Fayek, and X. Chang, \u201cSWAP-NAS: Sample-wise activation patterns for ultra-fast NAS,\u201d Proc. ICLR, pp.1-26, 2024."},{"key":"49","unstructured":"[49] H. Xiong, L. Huang, M. Yu, L. Liu, F. Zhu, and L. Shao, \u201cOn the number of linear regions of convolutional neural networks,\u201d Proc. ICML, pp.10514-10523, 2020."},{"key":"50","unstructured":"[50] S. Arora, S.S. Du, Z. Li, R. Salakhutdinov, R. Wang, and D. Yu, \u201cHarnessing the power of infinitely wide deep nets on small-data tasks,\u201d Proc. ICLR, pp.1-13, 2020."},{"key":"51","doi-asserted-by":"crossref","unstructured":"[51] J. Lee, S. Schoenholz, J. Pennington, B. Adlam, L. Xiao, R. Novak, and J. Sohl-Dickstein, \u201cFinite versus infinite neural networks: An empirical study,\u201d Proc. NeurIPS, pp.15156-15172, 2020.","DOI":"10.1088\/1742-5468\/abc62b"},{"key":"52","unstructured":"[52] M. Seleznova and G. Kutyniok, \u201cAnalyzing finite neural networks: Can we trust neural tangent kernel theory?,\u201d Proc. MSML, pp.868-895, 2022."},{"key":"53","unstructured":"[53] Z. Zhu, F. Liu, G. Chrysos, and V. Cevher, \u201cGeneralization properties of NAS under activation and skip connection search,\u201d Proc. NeurIPS, pp.23551-23565, 2022."},{"key":"54","unstructured":"[54] E. Jang, S. Gu, and B. Poole, \u201cCategorical reparameterization with Gumbel-Softmax,\u201d Proc. ICLR, 2017."},{"key":"55","unstructured":"[55] C.J. Maddison, A. Mnih, and Y.W. Teh, \u201cThe concrete distribution: A continuous relaxation of discrete random variables,\u201d Proc. ICLR, 2017."},{"key":"56","unstructured":"[56] Y. Shu, Z. Dai, Z. Wu, and B.K.H. Low, \u201cUnifying and boosting gradient-based training-free neural architecture search,\u201d Proc. NeurIPS, pp.33001-33015, 2022."},{"key":"57","unstructured":"[57] J. Snoek, H. Larochelle, and R.P. Adams, \u201cPractical Bayesian optimization of machine learning algorithms,\u201d Proc. NeurIPS, pp.1-9, 2012."},{"key":"58","doi-asserted-by":"crossref","unstructured":"[58] K. He, X. Zhang, S. Ren, and J. Sun, \u201cDelving deep into rectifiers: Surpassing human-level performance on ImageNet classification,\u201d Proc. ICCV, pp.1026-1034, 2015. 10.1109\/ICCV.2015.123","DOI":"10.1109\/ICCV.2015.123"},{"key":"59","doi-asserted-by":"crossref","unstructured":"[59] J. Deng, W. Dong, R. Socher, L.J. Li, K. Li, and L. Fei-Fei, \u201cImageNet: A large-scale hierarchical image database,\u201d Proc. CVPR, pp.248-255, 2009. 10.1109\/CVPR.2009.5206848","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"60","unstructured":"[60] P. Chrabaszcz, I. Loshchilov, and F. Hutter, \u201cA downsampled variant of ImageNet as an alternative to the CIFAR datasets,\u201d arXiv:1707.08819, pp.1-9, 2017. 10.48550\/arXiv.1707.08819"},{"key":"61","unstructured":"[61] W. Chen, W. Huang, X. Du, X. Song, Z. Wang, and D. Zhou, \u201cAuto-scaling vision transformers without training,\u201d Proc. ICLR, pp.1-14, 2022."},{"key":"62","doi-asserted-by":"publisher","unstructured":"[62] K.T. Chitty-Venkata, M. Emani, V. Vishwanath, and A.K. Somani, \u201cNeural architecture search for transformers: A survey,\u201d IEEE Access, vol.10, pp.108374-108412, 2022. 10.1109\/ACCESS.2022.3212767","DOI":"10.1109\/ACCESS.2022.3212767"},{"key":"63","unstructured":"[63] B. Hanin, \u201cWhich neural net architectures give rise to exploding and vanishing gradients?,\u201d Proc. NeurIPS, pp.1-10, 2018."},{"key":"64","doi-asserted-by":"publisher","unstructured":"[64] B. Hanin and M. Nica, \u201cProducts of many large random matrices and gradients in deep neural networks,\u201d Communications in Mathematical Physics, vol.376, no.1, pp.287-322, 2020. 10.1007\/s00220-019-03624-z","DOI":"10.1007\/s00220-019-03624-z"},{"key":"65","unstructured":"[65] S. Fort, G.K. Dziugaite, M. Paul, S. Kharaghani, D.M. Roy, and S. Ganguli, \u201cDeep learning versus kernel learning: An empirical study of loss landscape geometry and the time evolution of the neural tangent kernel,\u201d Proc. NeurIPS, pp.5850-5861, 2020."},{"key":"66","doi-asserted-by":"crossref","unstructured":"[66] J. Mok, B. Na, J.-H. Kim, D. Han, and S. Yoon, \u201cDemystifying the neural tangent kernel from a practical perspective: Can it be trusted for neural architecture search without training?,\u201d Proc. CVPR, pp.11861-11870, 2022. 10.1109\/CVPR52688.2022.01156","DOI":"10.1109\/CVPR52688.2022.01156"}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E108.D\/10\/E108.D_2024EDP7245\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,4]],"date-time":"2025-10-04T03:28:32Z","timestamp":1759548512000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E108.D\/10\/E108.D_2024EDP7245\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,1]]},"references-count":66,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2025]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2024edp7245","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"value":"0916-8532","type":"print"},{"value":"1745-1361","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,10,1]]},"article-number":"2024EDP7245"}}