{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T17:19:18Z","timestamp":1777569558658,"version":"3.51.4"},"reference-count":46,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,4,6]],"date-time":"2025-04-06T00:00:00Z","timestamp":1743897600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,4,6]],"date-time":"2025-04-06T00:00:00Z","timestamp":1743897600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,4,6]]},"DOI":"10.1109\/icassp49660.2025.10887793","type":"proceedings-article","created":{"date-parts":[[2025,3,12]],"date-time":"2025-03-12T17:15:19Z","timestamp":1741799719000},"page":"1-5","source":"Crossref","is-referenced-by-count":1,"title":["Full-Rank No More: Low-Rank Weight Training for Modern Speech Recognition Models"],"prefix":"10.1109","author":[{"given":"Adriana","family":"Fernandez-Lopez","sequence":"first","affiliation":[{"name":"Meta,UK"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shiwei","family":"Liu","sequence":"additional","affiliation":[{"name":"University of Oxford,UK"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lu","family":"Yin","sequence":"additional","affiliation":[{"name":"University of Surrey,UK"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Stavros","family":"Petridis","sequence":"additional","affiliation":[{"name":"Meta,UK"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Maja","family":"Pantic","sequence":"additional","affiliation":[{"name":"Meta,UK"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","first-page":"2","article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","volume-title":"Proceedings of naacL-HLT","volume":"1","author":"Devlin"},{"key":"ref2","article-title":"Gpt-4 technical report","author":"Achiam","year":"2023"},{"key":"ref3","article-title":"Llama 2: Open foundation and fine-tuned chat models","author":"Touvron","year":"2023"},{"issue":"120","key":"ref4","first-page":"1","article-title":"Switch transformers: Scaling to trillion parameter models with simple and efficient sparsity","volume":"23","author":"Fedus","year":"2022","journal-title":"Journal of Machine Learning Research"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2618"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747729"},{"key":"ref7","article-title":"Deep compression: Compressing deep neural networks with pruning, trained quantization and huffman coding","volume-title":"Proceedings of ICLR","author":"Han"},{"key":"ref8","first-page":"6989","article-title":"Do we actually need dense over-parameterization? in-time over-parameterization in sparse training","volume-title":"ICML","author":"Molchanov"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-1153"},{"key":"ref10","article-title":"Dynamic sparsity is channel-level sparsity learner","volume-title":"Proceedings of NeurIPS","volume":"36","author":"Yin"},{"key":"ref11","article-title":"LoRA: Low-rank adaptation of large language models","volume-title":"Proceedings of ICLR","author":"Hu"},{"key":"ref12","article-title":"Adalora: Adaptive budget allocation for parameter-efficient fine-tuning","volume-title":"ICLR","author":"Zhang"},{"key":"ref13","first-page":"79320","article-title":"Controlling text-to-image diffusion by orthogonal finetuning","volume-title":"Proceedings of NeurIPS","volume":"36","author":"Qiu"},{"key":"ref14","article-title":"Dora: Weight-decomposed low-rank adaptation","volume-title":"Proceedings of ICML","author":"Liu"},{"key":"ref15","article-title":"Owlore: Outlier-weighed layerwise sampled low-rank projection for memory-efficient llm fine-tuning","author":"Li","year":"2024"},{"key":"ref16","article-title":"Adarank: Disagreement based module rank prediction for low-rank adaptation","author":"Dong","year":"2024"},{"key":"ref17","article-title":"Investigating low-rank training in transformer language models: Efficiency and scaling analysis","volume-title":"Proceedings of ICML","author":"Wei"},{"key":"ref18","article-title":"Relora: High-rank training through low-rank updates","volume-title":"Proceedings of ICLR","author":"Lialin"},{"key":"ref19","first-page":"578","article-title":"Cuttlefish: Low-rank model training without all the tuning","volume-title":"Proceedings of MLSys","volume":"5","author":"Wang"},{"key":"ref20","article-title":"From galore to welore: Memory-efficient finetuning with adaptive low-rank weight projection","author":"Jaiswal","year":"2024"},{"key":"ref21","article-title":"Galore: Memory-efficient llm training by gradient low-rank projection","volume-title":"Proceedings of ICML","author":"Zhao"},{"key":"ref22","article-title":"Q-galore: Quantized galore with int4 projection and layer-adaptive low-rank gradients","author":"Zhang","year":"2024"},{"key":"ref23","article-title":"SLTrain: a sparse plus lowrank approach for parameter and memory efficient pretraining","volume-title":"Proceedings of NeurIPS","author":"Han"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2013-552"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053878"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1417"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-892"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095006"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472820"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096889"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"ref32","first-page":"249","article-title":"Understanding the difficulty of training deep feedforward neural networks","volume-title":"Proceedings of AISTATS. JMLR Workshop and Conference Proceedings","author":"Glorot"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.123"},{"key":"ref34","article-title":"On the initialisation of wide lowrank feedforward neural networks","author":"Saada","year":"2023"},{"key":"ref35","article-title":"Initialization and regularization of factorized neural layers","volume-title":"Proceedings of ICLR","author":"Khodak"},{"key":"ref36","article-title":"Training cnns with low-rank filters for efficient image classification","volume-title":"Proceedings of ICLR","author":"Ioannou"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"ref38","article-title":"LRS3TED: a large-scale dataset for visual speech recognition","author":"Afouras","year":"2018"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-462"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1929"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1145\/3197517.3201357"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1038\/s42256-022-00550-z"},{"key":"ref43","first-page":"7613","article-title":"End-to-end audiovisual speech recognition with conformers","volume-title":"ICASSP","author":"Ma"},{"key":"ref44","article-title":"Decoupled Weight Decay Regularization","volume-title":"ICLR","author":"Loshchilov"},{"key":"ref45","article-title":"Scaling laws and computeoptimal training beyond fixed training durations","volume-title":"Workshop on Efficient Systems for Foundation Models II, in ICML","author":"H\u00f6gele"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1016\/0167-6393(93)90095-3"}],"event":{"name":"ICASSP 2025 - 2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","location":"Hyderabad, India","start":{"date-parts":[[2025,4,6]]},"end":{"date-parts":[[2025,4,11]]}},"container-title":["ICASSP 2025 - 2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10887540\/10887541\/10887793.pdf?arnumber=10887793","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T05:24:28Z","timestamp":1774416268000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10887793\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,4,6]]},"references-count":46,"URL":"https:\/\/doi.org\/10.1109\/icassp49660.2025.10887793","relation":{},"subject":[],"published":{"date-parts":[[2025,4,6]]}}}