{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,7]],"date-time":"2025-11-07T18:20:29Z","timestamp":1762539629459,"version":"build-2065373602"},"reference-count":68,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U2336211"],"award-info":[{"award-number":["U2336211"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62576364"],"award-info":[{"award-number":["62576364"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Start-up Grant of the City University of Hong Kong","award":["9382010"],"award-info":[{"award-number":["9382010"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. on Image Process."],"published-print":{"date-parts":[[2025]]},"DOI":"10.1109\/tip.2025.3624968","type":"journal-article","created":{"date-parts":[[2025,11,3]],"date-time":"2025-11-03T18:47:17Z","timestamp":1762195637000},"page":"7109-7122","source":"Crossref","is-referenced-by-count":0,"title":["ScaleNet: Scaling up Pretrained Neural Networks With Incremental Parameters"],"prefix":"10.1109","volume":"34","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6237-7028","authenticated-orcid":false,"given":"Zhiwei","family":"Hao","sequence":"first","affiliation":[{"name":"School of Information and Electronics, Beijing Institute of Technology, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianyuan","family":"Guo","sequence":"additional","affiliation":[{"name":"Department of Computer Science, City University of Hong Kong, Hong Kong, SAR, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5659-3464","authenticated-orcid":false,"given":"Li","family":"Shen","sequence":"additional","affiliation":[{"name":"School of Cyber Science and Technology, Sun Yat-sen University, Shenzhen Campus, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9761-2702","authenticated-orcid":false,"given":"Kai","family":"Han","sequence":"additional","affiliation":[{"name":"Huawei Noah&#x2019;s Ark Lab, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yehui","family":"Tang","sequence":"additional","affiliation":[{"name":"Huawei Noah&#x2019;s Ark Lab, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7532-0496","authenticated-orcid":false,"given":"Han","family":"Hu","sequence":"additional","affiliation":[{"name":"School of Information and Electronics, Beijing Institute of Technology, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2709-4946","authenticated-orcid":false,"given":"Yunhe","family":"Wang","sequence":"additional","affiliation":[{"name":"Huawei Noah&#x2019;s Ark Lab, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","first-page":"1","article-title":"An image is worth 16\u00d716 words: Transformers for image recognition at scale","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Dosovitskiy"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"ref3","article-title":"Scaling laws for neural language models","author":"Kaplan","year":"2020","journal-title":"arXiv:2001.08361"},{"key":"ref4","article-title":"The llama 3 herd of models","author":"Grattafiori","year":"2024","journal-title":"arXiv:2407.21783"},{"key":"ref5","article-title":"Net2Net: Accelerating learning via knowledge transfer","author":"Chen","year":"2015","journal-title":"arXiv:1511.05641"},{"key":"ref6","first-page":"19893","article-title":"Staged training for transformer language models","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Shen"},{"key":"ref7","article-title":"On the transformer growth for progressive BERT training","author":"Gu","year":"2020","journal-title":"arXiv:2010.12562"},{"key":"ref8","article-title":"Bert2BERT: Towards reusable pretrained language models","author":"Chen","year":"2021","journal-title":"arXiv:2110.07143"},{"key":"ref9","first-page":"1","article-title":"Learning to grow pretrained models for efficient transformer training","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Wang"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.72"},{"key":"ref11","first-page":"2337","article-title":"Efficient training of BERT by progressively stacking","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Gong"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01216"},{"key":"ref13","article-title":"Learning implicitly recurrent CNNs through parameter sharing","author":"Savarese","year":"2019","journal-title":"arXiv:1902.09701"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01183"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref16","first-page":"10347","article-title":"Training data-efficient image transformers & distillation through attention","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Touvron"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/7503.003.0024"},{"key":"ref18","first-page":"524","article-title":"The cascade-correlation learning architecture","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"2","author":"Fahlman"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1162\/neco.2006.18.7.1527"},{"key":"ref20","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2018","journal-title":"arXiv:1810.04805"},{"issue":"143","key":"ref21","first-page":"18","article-title":"Generalization and network design strategies","volume":"19","author":"LeCun","year":"1989","journal-title":"Connectionism perspective"},{"key":"ref22","article-title":"ALBERT: A lite BERT for self-supervised learning of language representations","author":"Lan","year":"2019","journal-title":"arXiv:1909.11942"},{"key":"ref23","article-title":"Universal transformers","author":"Dehghani","year":"2018","journal-title":"arXiv:1807.03819"},{"key":"ref24","first-page":"688","article-title":"Deep equilibrium models","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Bai"},{"key":"ref25","first-page":"2229","article-title":"A tensorized transformer for language modeling","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Ma"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-emnlp.344"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.920"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6424"},{"key":"ref29","article-title":"Scaling down to scale up: A guide to parameter-efficient fine-tuning","author":"Lialin","year":"2023","journal-title":"arXiv:2303.15647"},{"key":"ref30","article-title":"Parameter-efficient transfer learning with diff pruning","author":"Guo","year":"2020","journal-title":"arXiv:2012.07463"},{"key":"ref31","first-page":"24193","article-title":"Training neural networks with fixed sparse masks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Sung"},{"key":"ref32","article-title":"Gradient-based parameter selection for efficient fine-tuning","author":"Zhang","year":"2023","journal-title":"arXiv:2312.10136"},{"key":"ref33","article-title":"BitFit: Simple parameter-efficient fine-tuning for transformer-based masked language-models","author":"Ben Zaken","year":"2021","journal-title":"arXiv:2106.10199"},{"key":"ref34","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Brown"},{"key":"ref35","article-title":"It\u2019s not just size that matters: Small language models are also few-shot learners","author":"Schick","year":"2020","journal-title":"arXiv:2009.07118"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.243"},{"key":"ref37","article-title":"Prefix-tuning: Optimizing continuous prompts for generation","author":"Lisa Li","year":"2021","journal-title":"arXiv:2101.00190"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"ref39","article-title":"Sequential modeling enables scalable learning for large vision models","author":"Bai","year":"2023","journal-title":"arXiv:2312.00785"},{"key":"ref40","article-title":"Data-efficient large vision models through sequential autoregression","author":"Guo","year":"2024","journal-title":"arXiv:2402.04841"},{"key":"ref41","first-page":"2790","article-title":"Parameter-efficient transfer learning for NLP","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Houlsby"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-demos.7"},{"key":"ref43","article-title":"Vision transformer adapter for dense predictions","author":"Chen","year":"2022","journal-title":"arXiv:2205.08534"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2024.3504252"},{"key":"ref45","article-title":"LoRA: Low-rank adaptation of large language models","author":"Hu","year":"2021","journal-title":"arXiv:2106.09685"},{"key":"ref46","article-title":"Layer normalization","author":"Ba","year":"2016","journal-title":"arXiv:1607.06450"},{"key":"ref47","article-title":"Multi-level residual networks from dynamical systems view","author":"Chang","year":"2017","journal-title":"arXiv:1710.10348"},{"key":"ref48","first-page":"2616","article-title":"Towards adaptive residual network training: A neural-ODE perspective","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Dong"},{"key":"ref49","article-title":"Towards a unified view of parameter-efficient transfer learning","author":"He","year":"2021","journal-title":"arXiv:2110.04366"},{"key":"ref50","article-title":"Test-time computing: From system-1 thinking to system-2 thinking","author":"Ji","year":"2025","journal-title":"arXiv:2501.02497"},{"key":"ref51","article-title":"On expressive power of looped transformers: Theoretical analysis and enhancement via timestep encoding","author":"Xu","year":"2024","journal-title":"arXiv:2410.01405"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.5555\/2969033.2969153"},{"key":"ref53","first-page":"1517","article-title":"Benefits of depth in neural networks","volume-title":"Proc. Conf. Learn. Theory","author":"Telgarsky"},{"key":"ref54","article-title":"Decoupled weight decay regularization","author":"Loshchilov","year":"2017","journal-title":"arXiv:1711.05101"},{"key":"ref55","first-page":"3519","article-title":"Similarity of neural network representations revisited","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Kornblith"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1906.07155"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.322"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.544"},{"volume-title":"MMSegmentation: Openmmlab Semantic Segmentation Toolbox and Benchmark","year":"2020","key":"ref60"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01228-1_26"},{"volume-title":"Stanford Alpaca: An Instruction-Following Llama Model","year":"2023","author":"Taori et al","key":"ref62"},{"key":"ref63","article-title":"BoolQ: Exploring the surprising difficulty of natural yes\/no questions","author":"Clark","year":"2019","journal-title":"arXiv:1905.10044"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6239"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1472"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1145\/3474381"},{"key":"ref67","article-title":"Think you have solved question answering? Try ARC, the AI2 reasoning challenge","author":"Clark","year":"2018","journal-title":"arXiv:1803.05457"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1260"}],"container-title":["IEEE Transactions on Image Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/83\/10795784\/11224410.pdf?arnumber=11224410","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,7]],"date-time":"2025-11-07T18:12:55Z","timestamp":1762539175000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11224410\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"references-count":68,"URL":"https:\/\/doi.org\/10.1109\/tip.2025.3624968","relation":{},"ISSN":["1057-7149","1941-0042"],"issn-type":[{"type":"print","value":"1057-7149"},{"type":"electronic","value":"1941-0042"}],"subject":[],"published":{"date-parts":[[2025]]}}}