{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T13:58:24Z","timestamp":1783605504130,"version":"3.55.0"},"reference-count":55,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"6","license":[{"start":{"date-parts":[[2024,9,1]],"date-time":"2024-09-01T00:00:00Z","timestamp":1725148800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,9,1]],"date-time":"2024-09-01T00:00:00Z","timestamp":1725148800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,9,1]],"date-time":"2024-09-01T00:00:00Z","timestamp":1725148800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE J. Sel. Top. Signal Process."],"published-print":{"date-parts":[[2024,9]]},"DOI":"10.1109\/jstsp.2024.3431927","type":"journal-article","created":{"date-parts":[[2024,7,22]],"date-time":"2024-07-22T17:59:09Z","timestamp":1721671149000},"page":"1059-1069","source":"Crossref","is-referenced-by-count":4,"title":["One is Not Enough: Parameter-Efficient Fine-Tuning With Multiplicative Sparse Factorization"],"prefix":"10.1109","volume":"18","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5451-2704","authenticated-orcid":false,"given":"Xuxi","family":"Chen","sequence":"first","affiliation":[{"name":"Department of Electrical and Computer Engineering, UT Austin, Austin, TX, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7774-8197","authenticated-orcid":false,"given":"Tianlong","family":"Chen","sequence":"additional","affiliation":[{"name":"Department of Computer Science, The University of North Carolina at Chapel Hill, Chapel Hill, NC, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yu","family":"Cheng","sequence":"additional","affiliation":[{"name":"The Chinese University of Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Weizhu","family":"Chen","sequence":"additional","affiliation":[{"name":"Microsoft Research, Redmond, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ahmed Hassan","family":"Awadallah","sequence":"additional","affiliation":[{"name":"Microsoft Research, Redmond, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2050-5693","authenticated-orcid":false,"given":"Zhangyang","family":"Wang","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering, UT Austin, Austin, TX, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"issue":"8","key":"ref1","first-page":"9","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI Blog"},{"key":"ref2","first-page":"2790","article-title":"Parameter-efficient transfer learning for NLP","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Houlsby","year":"2019"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.353"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.243"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.378"},{"key":"ref6","article-title":"LoRA: Low-rank adaptation of large language models","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Hu","year":"2022"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2022.3178101"},{"key":"ref8","article-title":"Measuring the intrinsic dimension of objective landscapes","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Li","year":"2018"},{"key":"ref9","article-title":"Improving neural network training in low dimensional random bases","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Gressmann","year":"2020"},{"key":"ref10","article-title":"Towards a unified view of parameter-efficient transfer learning","volume-title":"Proc. Int. Conf. Learn. Representationsc","author":"He","year":"2022"},{"key":"ref11","article-title":"What would elsa do? Freezing layers during transformer fine-tuning","author":"Lee","year":"2019"},{"key":"ref12","volume-title":"Statistical Theory of Extreme Values and Some Practical Applications: A Series of Lectures","volume":"33","author":"Gumbel","year":"1954"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.7713\/ijms.2013.0032"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref15","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Dosovitskiy","year":"2018"},{"key":"ref16","first-page":"10347","article-title":"Training data-efficient image transformers & distillation through attention","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Touvron","year":"2021"},{"key":"ref17","article-title":"Chasing sparsity in vision transformers: An end-to-end exploration","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Chen","year":"2021"},{"key":"ref18","article-title":"Anti-oversmoothing in deep vision transformers via the fourier domain analysis: From theory to practice","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Wang","year":"2022"},{"key":"ref19","article-title":"Delayed propagation transformer: A universal computation engine towards practical control in cyber-physical systems","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zheng","year":"2021"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1810.04805"},{"key":"ref21","article-title":"RoBERTa: A robustly optimized BERT pretraining approach","author":"Liu","year":"2019"},{"key":"ref22","article-title":"DeBERTa: Decoding-enhanced BERT with disentangled attention","volume-title":"Proc. Int. Conf. Learn. Representations","author":"He","year":"2020"},{"key":"ref23","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/p19-1355"},{"key":"ref25","first-page":"506","article-title":"Learning multiple visual domains with residual adapters","volume-title":"Proc. 31st Int. Conf. Neural Inf. Process. Syst.","author":"Rebuffi","year":"2017"},{"key":"ref26","first-page":"16664","article-title":"AdaptFormer: Adapting vision transformers for scalable visual recognition","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Chen","year":"2022"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.456"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i1.25187"},{"key":"ref29","article-title":"Sparse matrix factorization","author":"Neyshabur","year":"2013"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.173"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298681"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN.2017.7966420"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2019.2946636"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2015.04.071"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1016\/j.irfa.2016.10.009"},{"key":"ref36","first-page":"17413","article-title":"Scatterbrain: Unifying sparse and low-rank attention approximation","volume":"34","author":"Chen","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref37","article-title":"Pixelated butterfly: Simple and efficient sparse training for neural network models","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Chen","year":"2022"},{"key":"ref38","first-page":"4690","article-title":"Monarch: Expressive structured matrices for efficient and accurate training","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Dao","year":"2022"},{"key":"ref39","first-page":"6","article-title":"Overview of mini-batch gradient descent","volume":"575","author":"Hinton","year":"2012","journal-title":"Neural Netw. Mach. Learn."},{"key":"ref40","article-title":"Decoupled weight decay regularization","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Loshchilov","year":"2019"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.541"},{"key":"ref42","article-title":"Rethinking the smaller-norm-less-informative assumption in channel pruning of convolution layers","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Ye","year":"2018"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00958"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2021.107899"},{"key":"ref45","article-title":"Understanding straight-through estimator in training activation quantized neural nets","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Yin","year":"2019"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D13-1170"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/n18-1101"},{"key":"ref48","first-page":"1797","article-title":"Don\u2019t give me the details, just the summary! Topic-aware convolutional neural networks for extreme summarization","volume-title":"Proc. 2018 Conf. Empir. Methods Natural Lang. Process.","author":"Narayan","year":"2018"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W18-5446"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.4324\/9781003022022-6"},{"key":"ref51","first-page":"74","article-title":"ROUGE: A package for automatic evaluation of summaries","volume-title":"Proc. Text Summarization Branches Out","author":"Lin","year":"2004"},{"key":"ref52","article-title":"P-tuning v2: Prompt tuning can be comparable to fine-tuning universally across scales and tasks","volume-title":"Proc. 60th Annu. Meeting Assoc. Computat. Linguistics","author":"Liu","year":"2021"},{"key":"ref53","first-page":"3469","article-title":"Learn-to-share: A hardware-friendly transfer learning framework exploiting computation and parameter sharing","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Fu","year":"2021"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-short.1"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1016\/j.aiopen.2023.08.012"}],"container-title":["IEEE Journal of Selected Topics in Signal Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/4200690\/10852353\/10606314.pdf?arnumber=10606314","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,6]],"date-time":"2025-02-06T18:40:32Z","timestamp":1738867232000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10606314\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,9]]},"references-count":55,"journal-issue":{"issue":"6"},"URL":"https:\/\/doi.org\/10.1109\/jstsp.2024.3431927","relation":{},"ISSN":["1932-4553","1941-0484"],"issn-type":[{"value":"1932-4553","type":"print"},{"value":"1941-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,9]]}}}