{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,13]],"date-time":"2026-04-13T00:45:00Z","timestamp":1776041100910,"version":"3.50.1"},"reference-count":106,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2024,8,19]],"date-time":"2024-08-19T00:00:00Z","timestamp":1724025600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,8,19]],"date-time":"2024-08-19T00:00:00Z","timestamp":1724025600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2024,9]]},"DOI":"10.1007\/s10772-024-10121-9","type":"journal-article","created":{"date-parts":[[2024,8,19]],"date-time":"2024-08-19T14:03:08Z","timestamp":1724076188000},"page":"765-779","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Efficiency-oriented approaches for self-supervised speech representation learning"],"prefix":"10.1007","volume":"27","author":[{"given":"Luis","family":"Lugo","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Valentin","family":"Vielzeuf","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,8,19]]},"reference":[{"key":"10121_CR1","volume-title":"Proceedings of the 5th workshop on research in computational linguistic typology and multilingual","author":"BM Abdullah","year":"2023","unstructured":"Abdullah, B. M., Shaik, M. M., & Klakow, D. (2023). On the nature of discrete speech representations in multilingual self-supervised models. In\u00a0Proceedings of the 5th workshop on research in computational linguistic typology and multilingual. NLP."},{"key":"10121_CR2","unstructured":"Allen-Zhu, Z., & Li, Y. (2020). Towards understanding ensemble, knowledge distillation and self-distillation in deep learning. arXiv preprint arXiv:2012.09816"},{"key":"10121_CR3","volume-title":"International conference on machine learning","author":"D Amodei","year":"2016","unstructured":"Amodei, D., Ananthanarayanan, S., Anubhai, R., Bai, J., Battenberg, E., Case, C., Casper, J., Catanzaro, B., Cheng, Q., Chen, G., Chen, J., Chen, J., Chen, Z., Chrzanowski, M., Coates, A., Diamos, G., Ding, K., Du, N., Elsen, E., \u2026 Zhu, Z. (2016). Deep speech 2: End-to-end speech recognition in English and Mandarin. In\u00a0International conference on machine learning (ICML). PMLR."},{"key":"10121_CR4","doi-asserted-by":"crossref","unstructured":"Arora, S., Dalmia, S., Denisov, P., Chang, X., Ueda, Y., Peng, Y., Zhang, Y., Kumar, S., Ganesan, K., Yan, B., Vu, N., Black, A., & Watanabe, S. (2022). Espnet-slu: Advancing spoken language understanding through ESPnet. In IEEE international conference on acoustics, speech and signal processing (ICASSP).","DOI":"10.1109\/ICASSP43922.2022.9747674"},{"key":"10121_CR5","volume-title":"Deep versus wide: An analysis of student architectures for task-agnostic knowledge distillation of self-supervised speech models","author":"T Ashihara","year":"2022","unstructured":"Ashihara, T., Moriya, T., Matsuura, K., & Tanaka, T. (2022). Deep versus wide: An analysis of student architectures for task-agnostic knowledge distillation of self-supervised speech models.\u00a0In Interspeech."},{"key":"10121_CR6","doi-asserted-by":"crossref","unstructured":"Babu, A., Wang, C., Tjandra, A., Lakhotia, K., Xu, Q., Goyal, N., Singh, K., Platen, P., Saraf, Y., Pino, J., Baevski, A., Conneau, A., & Auli, M. (2021). Xls-r: Self-supervised cross-lingual speech representation learning at scale. In Interspeech.","DOI":"10.21437\/Interspeech.2022-143"},{"key":"10121_CR7","unstructured":"Baevski, A., Schneider, S., & Auli, M. (2020). vq-wav2vec: Self-supervised learning of discrete speech representations. In International conference on learning representations (ICLR)."},{"key":"10121_CR8","unstructured":"Baevski, A., Zhou, Y., Mohamed, A., & Auli, M. (2020). wav2vec 2.0: A framework for self-supervised learning of speech representations. In Advances in neural information processing systems (NIPS)."},{"key":"10121_CR9","volume-title":"International conference on machine learning","author":"A Baevski","year":"2023","unstructured":"Baevski, A., Babu, A., Hsu, W.-N., & Auli, M. (2023). Efficient self-supervised learning with contextualized target representations for vision, speech and language. In\u00a0International conference on machine learning (ICML). PMLR."},{"key":"10121_CR10","volume-title":"International conference on machine learning","author":"A Baevski","year":"2022","unstructured":"Baevski, A., Hsu, W.-N., Xu, Q., Babu, A., Gu, J., & Auli, M. (2022). Data2vec: A general framework for self-supervised learning in speech, vision and language. In\u00a0International conference on machine learning (ICML). PMLR."},{"key":"10121_CR11","doi-asserted-by":"crossref","unstructured":"Bartley, T.M., Jia, F., Puvvada, K.C., Kriman, S., & Ginsburg, B. (2023). Accidental learners: Spoken language identification in multilingual self-supervised models. In IEEE international conference on acoustics, speech and signal processing (ICASSP).","DOI":"10.1109\/ICASSP49357.2023.10096407"},{"key":"10121_CR12","doi-asserted-by":"crossref","unstructured":"Bello, I., Zoph, B., Le, Q., Vaswani, A., & Shlens, J. (2019). Attention augmented convolutional networks. In Proceedings of the IEEE\/CVF international conference on computer vision.","DOI":"10.1109\/ICCV.2019.00338"},{"key":"10121_CR13","unstructured":"Brown, T., Mann, B., Ryder, N., Subbiah, M., Kaplan, J.D., Dhariwal, P., Neelakantan, A., Shyam, P., Sastry, G., Askell, A., Agarwal, S., Herbert-Voss, A., Krueger, G., Henighan, T., Child, R., Ramesh, A., Ziegler, D.M., Wu, J., Winter, C., Hesse, C., Chen, M., Sigler, E., Litwin, M., Gray, S., Chess, B., Clark, J., Berner, C., McCandlish, S., Radford, A., Sutskever, I., Amodei, D. (2020). Language models are few-shot learners. In Advances in neural information processing systems (NIPS)."},{"key":"10121_CR14","doi-asserted-by":"crossref","unstructured":"Chang, H.-J., Yang, S.-w., & Lee, H.-y. (2022). Distilhubert: Speech representation learning by layer-wise distillation of hidden-unit Bert. In IEEE international conference on acoustics, speech and signal processing (ICASSP).","DOI":"10.1109\/ICASSP43922.2022.9747490"},{"key":"10121_CR15","volume-title":"Reducing barriers to self-supervised learning: Hubert pre-training with academic compute","author":"W Chen","year":"2023","unstructured":"Chen, W., Chang, X., Peng, Y., Ni, Z., Maiti, S., & Watanabe, S. (2023). Reducing barriers to self-supervised learning: Hubert pre-training with academic compute. In Interspeech."},{"key":"10121_CR16","volume-title":"International conference on machine learning","author":"T Chen","year":"2020","unstructured":"Chen, T., Kornblith, S., Norouzi, M., & Hinton, G. (2020). A simple framework for contrastive learning of visual representations. In\u00a0International conference on machine learning (ICML). PMLR."},{"issue":"6","key":"10121_CR17","doi-asserted-by":"publisher","first-page":"1505","DOI":"10.1109\/JSTSP.2022.3188113","volume":"16","author":"S Chen","year":"2022","unstructured":"Chen, S., Wang, C., Chen, Z., Wu, Y., Liu, S., Chen, Z., Li, J., Kanda, N., Yoshioka, T., Xiao, X., Wu, J., Zhou, L., Ren, S., Qian, Y., Qian, Y., Wu, J., Zeng, M., Yu, X., Wei, F. (2022). Wavlm: Large-scale self-supervised pre-training for full stack speech processing. IEEE Journal of Selected Topics in Signal Processing, 16(6), 1505\u20131518.","journal-title":"IEEE Journal of Selected Topics in Signal Processing"},{"issue":"6","key":"10121_CR18","doi-asserted-by":"publisher","first-page":"406","DOI":"10.1250\/ast.40.406","volume":"40","author":"Y Chiba","year":"2019","unstructured":"Chiba, Y., Nose, T., & Ito, A. (2019). Multi-condition training for noise-robust speech emotion recognition. Acoustical Science and Technology, 40(6), 406\u2013409.","journal-title":"Acoustical Science and Technology"},{"key":"10121_CR19","unstructured":"Child, R., Gray, S., Radford, A., & Sutskever, I. (2019). Generating long sequences with sparse transformers. arXiv preprint arXiv:1904.10509"},{"key":"10121_CR20","unstructured":"Choromanski, K. M., Likhosherstov, V., Dohan, D., Song, X., Gane, A., Sarlos, T., Hawkins, P., Davis, J. Q., Mohiuddin, A., Kaiser, L., Belanger, D. B., Colwell, L. J., & Weller, A. (2021). Rethinking attention with performers. In International conference on learning representations (ICLR)."},{"key":"10121_CR21","volume-title":"Speech2vec: A sequence-to-sequence framework for learning word embeddings from speech","author":"Y-A Chung","year":"2018","unstructured":"Chung, Y.-A., & Glass, J. (2018). Speech2vec: A sequence-to-sequence framework for learning word embeddings from speech. In Interspeech."},{"key":"10121_CR22","doi-asserted-by":"crossref","unstructured":"Conneau, A., Baevski, A., Collobert, R., Mohamed, A., & Auli, M. (2021). Unsupervised cross-lingual representation learning for speech recognition. In Interspeech.","DOI":"10.21437\/Interspeech.2021-329"},{"key":"10121_CR23","unstructured":"Dao, T., Fu, D., Ermon, S., Rudra, A., & R\u00e9, C. (2022). Flashattention: Fast and memory-efficient exact attention with io-awareness. In Advances in neural information processing systems (NIPS)."},{"key":"10121_CR24","unstructured":"Dao, T., Fu, D.Y., Saab, K.K., Thomas, A.W., Rudra, A., & R\u00e9, C. (2022). Hungry hungry hippos: Towards language modeling with state space models. arXiv preprint arXiv:2212.14052"},{"key":"10121_CR25","unstructured":"Devlin, J., Chang, M.-W., Lee, K., & Toutanova, K. (2019). BERT: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of the conference of the North American Chapter of the Association for computational linguistics: Human language technologies."},{"key":"10121_CR26","volume-title":"The zero resource speech challenge 2021: Spoken language modelling","author":"E Dunbar","year":"2021","unstructured":"Dunbar, E., Bernard, M., Hamilakis, N., Nguyen, T. A., Seyssel, M., Roz\u00e9, P., Rivi\u00e8re, M., Kharitonov, E., & Dupoux, E. (2021). The zero resource speech challenge 2021: Spoken language modelling. In Interspeech."},{"issue":"3","key":"10121_CR27","doi-asserted-by":"publisher","first-page":"42","DOI":"10.1109\/MSP.2021.3134634","volume":"39","author":"L Ericsson","year":"2022","unstructured":"Ericsson, L., Gouk, H., Loy, C. C., & Hospedales, T. M. (2022). Self-supervised representation learning: Introduction, advances, and challenges. IEEE Signal Processing Magazine, 39(3), 42\u201362.","journal-title":"IEEE Signal Processing Magazine"},{"key":"10121_CR28","volume-title":"Lebenchmark: A reproducible framework for assessing self-supervised representation learning from speech","author":"S Evain","year":"2021","unstructured":"Evain, S., Nguyen, H., Le, H., Boito, M. Z., Mdhaffar, S., Alisamir, S., Tong, Z., Tomashenko, N., Dinarelli, M., Parcollet, T., Allauzen, A., Esteve, Y., Lecouteux, B., Portet, F., Rossato, S., Ringeval, F., Schwab, D., & Besacier, L. (2021). LeBenchmark: A reproducible framework for assessing self-supervised representation learning from speech. In Interspeech."},{"key":"10121_CR29","doi-asserted-by":"crossref","unstructured":"Gao, Y., Fernandez-Marques, J., Parcollet, T., Mehrotra, A., & Lane, N. D. (2022). Federated self-supervised speech representations: Are we there yet? arXiv preprint arXiv:2204.02804","DOI":"10.21437\/Interspeech.2022-10644"},{"key":"10121_CR30","doi-asserted-by":"crossref","unstructured":"Gaol, Y., Fernandez-Marques, J., Parcollet, T., Gusmao, P. P., & Lane, N. D. (2023). Match to win: Analysing sequences lengths for efficient self-supervised learning in speech and audio. In IEEE spoken language technology workshop (SLT)","DOI":"10.1109\/SLT54892.2023.10023410"},{"key":"10121_CR31","first-page":"1","volume":"385","author":"A Graves","year":"2012","unstructured":"Graves, A. (2012). Supervised sequence labelling with recurrent neural networks. Studies in Computational Intelligence, 385, 1\u2013131.","journal-title":"Studies in Computational Intelligence"},{"key":"10121_CR32","unstructured":"Grill, J.-B., Strub, F., Altch\u00e9, F., Tallec, C., Richemond, P., Buchatskaya, E., Doersch, C., Avila Pires, B., Guo, Z., Gheshlaghi Azar, M., Piot, B., Kavukcuoglu, K., Munos, R., & Valko, M. (2020). Bootstrap your own latent: A new approach to self-supervised learning. In Advances in neural information processing systems (NIPS)"},{"key":"10121_CR33","doi-asserted-by":"crossref","unstructured":"Gulati, A., Qin, J., Chiu, C.-C., Parmar, N., Zhang, Y., Yu, J., Han, W., Wang, S., Zhang, Z., Wu, Y., & Pang, R. (2020). Conformer: Convolution-augmented transformer for speech recognition. In Interspeech.","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"10121_CR34","doi-asserted-by":"publisher","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","volume":"29","author":"W-N Hsu","year":"2021","unstructured":"Hsu, W.-N., Bolte, B., Tsai, Y.-H.H., Lakhotia, K., Salakhutdinov, R., & Mohamed, A. (2021). Hubert: Self-supervised speech representation learning by masked prediction of hidden units. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 29, 3451\u20133460.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"10121_CR35","unstructured":"Hu, E. J., Shen, Y., Wallis, P., Allen-Zhu, Z., Li, Y., Wang, S., Wang, L., & Chen, W. (2021). Lora: Low-rank adaptation of large language models. In International conference on learning representations (ICLR)."},{"key":"10121_CR36","unstructured":"Huang, W., Zhang, Z., Yeung, Y. T., Jiang, X., & Liu, Q. (2022). Spiral: Self-supervised perturbation-invariant representation learning for speech pre-training. In International conference on learning representations (ICLR)."},{"key":"10121_CR37","unstructured":"Jang, E., Gu, S., & Poole, B. (2016). Categorical reparameterization with Gumbel\u2013softmax. In International conference on learning representations (ICLR)."},{"key":"10121_CR38","doi-asserted-by":"crossref","unstructured":"Kahn, J., Riviere, M., Zheng, W., Kharitonov, E., Xu, Q., Mazar\u00e9, P.-E., Karadayi, J., Liptchinsky, V., Collobert, R., Fuegen, C., Likhomanenko, T., Synnaeve, G., Joulin, A., Mohamed, A., & Dupoux, E. (2020). Librilight: A benchmark for ASR with limited or no supervision. In IEEE international conference on acoustics, speech and signal processing (ICASSP).","DOI":"10.1109\/ICASSP40776.2020.9052942"},{"key":"10121_CR39","unstructured":"Karimi Mahabadi, R., Henderson, J., & Ruder, S. (2021). Compacter: Efficient low-rank hypercomplex adapter layers. In Advances in neural information processing systems (NIPS)."},{"key":"10121_CR40","unstructured":"Kitaev, N., Kaiser, \u0141., & Levskaya, A. (2020). Reformer: The efficient transformer. arXiv preprint arXiv:2001.04451"},{"key":"10121_CR41","doi-asserted-by":"crossref","unstructured":"Lai, C.-I., Chuang, Y.-S., Lee, H.-Y., Li, S.-W., & Glass, J. (2021). Semi-supervised spoken language understanding via self-supervised speech and language model pretraining. In IEEE international conference on acoustics, speech and signal processing (ICASSP).","DOI":"10.1109\/ICASSP39728.2021.9414922"},{"key":"10121_CR42","unstructured":"Lai, C.-I.J., Zhang, Y., Liu, A.H., Chang, S., Liao, Y.-L., Chuang, Y.-S., Qian, K., Khurana, S., Cox, D., & Glass, J. (2021). Parp: Prune, adjust and re-prune for self-supervised speech recognition. In Advances in neural information processing systems (NIPS)."},{"key":"10121_CR43","doi-asserted-by":"crossref","unstructured":"Le, D., Zhang, X., Zheng, W., F\u00fcgen, C., Zweig, G., & Seltzer, M.L. (2019). From senones to Chenones: Tied context-dependent graphemes for hybrid speech recognition. In IEEE automatic speech recognition and understanding workshop (ASRU).","DOI":"10.1109\/ASRU46091.2019.9003972"},{"key":"10121_CR44","doi-asserted-by":"crossref","unstructured":"Lee, Y., & Jang, K., Goo, J., Jung, Y., & Kim, H.-R. (2022). Fithubert: Going thinner and deeper for knowledge distillation of speech self-supervised learning. In Interspeech.","DOI":"10.21437\/Interspeech.2022-11112"},{"key":"10121_CR45","doi-asserted-by":"crossref","unstructured":"Lee-Thorp, J., Ainslie, J., Eckstein, I., & Ontanon, S. (2022). Fnet: Mixing tokens with Fourier transforms. In Proceedings of the conference of the North American Chapter of the association for computational linguistics: Human language technologies.","DOI":"10.18653\/v1\/2022.naacl-main.319"},{"key":"10121_CR46","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.243","volume-title":"The power of scale for parameter-efficient prompt tuning","author":"B Lester","year":"2021","unstructured":"Lester, B., Al-Rfou, R., & Constant, N. (2021). The power of scale for parameter-efficient prompt tuning. EMNLP."},{"key":"10121_CR47","doi-asserted-by":"crossref","unstructured":"Li, X.L., & Liang, P. (2021). Prefix-tuning: Optimizing continuous prompts for generation. In Proceedings of the annual meeting of the association for computational linguistics and the international joint conference on natural language processing.","DOI":"10.18653\/v1\/2021.acl-long.353"},{"key":"10121_CR48","unstructured":"Lialin, V., Deshpande, V., & Rumshisky, A. (2023). Scaling down to scale up: A guide to parameter-efficient fine-tuning. arXiv preprint arXiv:2303.15647"},{"key":"10121_CR49","doi-asserted-by":"crossref","unstructured":"Lin, T.-Q., Lee, H.-y., & Tang, H. (2022). Melhubert: A simplified Hubert on Mel spectrogram. arXiv preprint arXiv:2211.09944","DOI":"10.1109\/ASRU57964.2023.10389700"},{"key":"10121_CR50","unstructured":"Liu, A. H., Chang, H.-J., Auli, M., Hsu, W.-N., & Glass, J. R. (2023). Dinosr: Self-distillation and online clustering for self-supervised speech representation learning. In Advances in neural information processing systems (NIPS)."},{"key":"10121_CR51","doi-asserted-by":"crossref","unstructured":"Maekawa, A., Kobayashi, N., Funakoshi, K., & Okumura, M. (2023). Dataset distillation with attention labels for fine-tuning BERT. In Proceedings of the 61st annual meeting of the association for computational linguistics.","DOI":"10.18653\/v1\/2023.acl-short.12"},{"key":"10121_CR52","unstructured":"Mehta, H., Gupta, A., Cutkosky, A., & Neyshabur, B. (2022). Long range language modeling via gated state spaces. arXiv preprint arXiv:2206.13947"},{"key":"10121_CR53","unstructured":"Micikevicius, P., Narang, S., Alben, J., Diamos, G., Elsen, E., Garcia, D., Ginsburg, B., Houston, M., Kuchaiev, O., Venkatesh, G., & Wu, H. (2018). Mixed precision training. In International conference on learning representations (ICLR)."},{"issue":"6","key":"10121_CR54","doi-asserted-by":"publisher","first-page":"1179","DOI":"10.1109\/JSTSP.2022.3207050","volume":"16","author":"A Mohamed","year":"2022","unstructured":"Mohamed, A., Lee, H.-Y., Borgholt, L., Havtorn, J. D., Edin, J., Igel, C., Kirchhoff, K., Li, S.-W., Livescu, K., Maal\u00f8e, L., Sainath, T. N., & Watanabe, S. (2022). Self-supervised speech representation learning: A review. IEEE Journal of Selected Topics in Signal Processing, 16(6), 1179\u2013210.","journal-title":"IEEE Journal of Selected Topics in Signal Processing"},{"key":"10121_CR55","doi-asserted-by":"crossref","unstructured":"Moumen, A., & Parcollet, T. (2023). Stabilising and accelerating light gated recurrent units for automatic speech recognition. In IEEE international conference on acoustics, speech and signal processing (ICASSP).","DOI":"10.1109\/ICASSP49357.2023.10095763"},{"key":"10121_CR56","unstructured":"Nguyen, T.A., Seyssel, M., Roz\u00e9, P., Rivi\u00e8re, M., Kharitonov, E., Baevski, A., Dunbar, E., & Dupoux, E. (2020). The zero resource speech benchmark 2021: Metrics and baselines for unsupervised spoken language modeling. In NeuRIPS workshop on self-supervised learning for speech and audio processing."},{"key":"10121_CR57","doi-asserted-by":"crossref","unstructured":"Panayotov, V., Chen, G., Povey, D., & Khudanpur, S. (2015). Librispeech: An ASR corpus based on public domain audio books. In IEEE international conference on acoustics, speech and signal processing (ICASSP).","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"10121_CR58","doi-asserted-by":"crossref","unstructured":"Parcollet, T., Dalen, R., Zhang, S., & Bhattacharya, S. (2023). Sumformer: A linear-complexity alternative to self-attention for speech recognition. arXiv preprint arXiv:2307.07421","DOI":"10.21437\/Interspeech.2024-40"},{"key":"10121_CR59","doi-asserted-by":"crossref","unstructured":"Parcollet, T., Zhang, S., Dalen, R., Ramos, A.G.C., & Bhattacharya, S. (2023). On the (in) efficiency of acoustic feature extractors for self-supervised speech representation learning. In Interspeech.","DOI":"10.21437\/Interspeech.2023-1510"},{"key":"10121_CR60","doi-asserted-by":"crossref","unstructured":"Park, D.S., Zhang, Y., Chiu, C.-C., Chen, Y., Li, B., Chan, W., Le, Q.V., & Wu, Y. (2020). Specaugment on large scale datasets. In IEEE international conference on acoustics, speech and signal processing (ICASSP).","DOI":"10.1109\/ICASSP40776.2020.9053205"},{"key":"10121_CR61","volume-title":"Specaugment: A simple data augmentation method for automatic speech recognition","author":"DS Park","year":"2019","unstructured":"Park, D. S., Chan, W., Zhang, Y., Chiu, C.-C., Zoph, B., Cubuk, E. D., & Le, Q. V. (2019). Specaugment: A simple data augmentation method for automatic speech recognition.\u00a0In Interspeech."},{"key":"10121_CR62","doi-asserted-by":"crossref","unstructured":"Pasad, A., Chou, J.-C., & Livescu, K. (2021). Layer-wise analysis of a self-supervised speech representation model. In IEEE automatic speech recognition and understanding workshop (ASRU).","DOI":"10.1109\/ASRU51503.2021.9688093"},{"key":"10121_CR63","doi-asserted-by":"crossref","unstructured":"Pasad, A., Shi, B., & Livescu, K. (2023). Comparative layer-wise analysis of self-supervised speech models. In IEEE international conference on acoustics, speech and signal processing (ICASSP).","DOI":"10.1109\/ICASSP49357.2023.10096149"},{"key":"10121_CR64","doi-asserted-by":"crossref","unstructured":"Peng, Y., Kim, K., Wu, F., Sridhar, P., & Watanabe, S. (2023). Structured pruning of self-supervised pre-trained models for speech recognition and understanding. In IEEE international conference on acoustics, speech and signal processing (ICASSP).","DOI":"10.1109\/ICASSP49357.2023.10095780"},{"key":"10121_CR65","volume-title":"Dphubert: Joint distillation and pruning of self-supervised speech models","author":"Y Peng","year":"2023","unstructured":"Peng, Y., Sudo, Y., Muhammad, S., & Watanabe, S. (2023). Dphubert: Joint distillation and pruning of self-supervised speech models. In Interspeech."},{"key":"10121_CR66","unstructured":"Poli, M., Massaroli, S., Nguyen, E., Fu, D.Y., Dao, T., Baccus, S., Bengio, Y., Ermon, S., & R\u00e9, C. (2023). Hyena hierarchy: Towards larger convolutional language models. arXiv preprint arXiv:2302.10866"},{"issue":"140","key":"10121_CR67","first-page":"1","volume":"21","author":"C Raffel","year":"2020","unstructured":"Raffel, C., Shazeer, N., Roberts, A., Lee, K., Narang, S., Matena, M., Zhou, Y., Li, W., & Liu, P. J. (2020). Exploring the limits of transfer learning with a unified text-to-text transformer. The Journal of Machine Learning Research, 21(140), 1\u201367.","journal-title":"The Journal of Machine Learning Research"},{"key":"10121_CR68","doi-asserted-by":"crossref","unstructured":"Ravanelli, M., Brakel, P., Omologo, M., & Bengio, Y. (2018). Light gated recurrent units for speech recognition. IEEE Transactions on Emerging Topics in Computational Intelligence,2","DOI":"10.1109\/TETCI.2017.2762739"},{"key":"10121_CR69","doi-asserted-by":"crossref","unstructured":"Reed, C.J., Yue, X., Nrusimha, A., Ebrahimi, S., Vijaykumar, V., Mao, R., Li, B., Zhang, S., Guillory, D., Metzger, S., Keutzer, K., & Darrell, T. (2022). Self-supervised pretraining improves self-supervised pretraining. In Proceedings of the IEEE\/CVF winter conference on applications of computer vision.","DOI":"10.1109\/WACV51458.2022.00112"},{"key":"10121_CR70","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00353","volume-title":"Efficient content-based sparse attention with routing transformers","author":"A Roy","year":"2021","unstructured":"Roy, A., Saffar, M., Vaswani, A., & Grangier, D. (2021). Efficient content-based sparse attention with routing transformers. Transactions of the Association for Computational Linguistics."},{"key":"10121_CR71","doi-asserted-by":"crossref","unstructured":"Sadhu, S., He, D., Huang, C.-W., Mallidi, S.H., Wu, M., Rastrow, A., Stolcke, A., Droppo, J., & Maas, R. (2021). Wav2vec-C: A self-supervised model for speech representation learning. In Interspeech.","DOI":"10.21437\/Interspeech.2021-717"},{"key":"10121_CR72","doi-asserted-by":"crossref","unstructured":"San, N., Bartelds, M., Browne, M., Clifford, L., Gibson, F., Mansfield, J., Nash, D., Simpson, J., Turpin, M., Vollmer, M., Wilmoth, S., & Jurafsky, D. (2021). Leveraging pre-trained representations to improve access to untranscribed speech from endangered languages. In IEEE automatic speech recognition and understanding workshop (ASRU)","DOI":"10.1109\/ASRU51503.2021.9688301"},{"key":"10121_CR73","unstructured":"Sanh, V., Debut, L., Chaumond, J., & Wolf, T. (2019). Distilbert, a distilled version of Bert: smaller, faster, cheaper and lighter. arXiv preprint arXiv:1910.01108"},{"key":"10121_CR74","first-page":"1","volume-title":"International conference on machine learning","author":"I Schlag","year":"2021","unstructured":"Schlag, I., Irie, K., & Schmidhuber, J. (2021). Linear transformers are secretly fast weight programmers. In\u00a0International conference on machine learning (ICML). PMLR."},{"key":"10121_CR75","doi-asserted-by":"crossref","unstructured":"Schneider, S., Baevski, A., Collobert, R., & Auli, M. (2019). wav2vec: Unsupervised pre-training for speech recognition. In Interspeech.","DOI":"10.21437\/Interspeech.2019-1873"},{"key":"10121_CR76","doi-asserted-by":"crossref","unstructured":"Seltzer, M.L., Yu, D., & Wang, Y. (2013). An investigation of deep neural networks for noise robust speech recognition. In IEEE international conference on acoustics, speech and signal processing (ICASSP).","DOI":"10.1109\/ICASSP.2013.6639100"},{"key":"10121_CR77","doi-asserted-by":"crossref","unstructured":"Seo, S., Kwak, D., & Lee, B. (2022). Integration of pre-trained networks with continuous token interface for end-to-end spoken language understanding. In IEEE international conference on acoustics, speech and signal processing (ICASSP).","DOI":"10.1109\/ICASSP43922.2022.9747047"},{"key":"10121_CR78","unstructured":"Shi, Y., Paige, B., Torr, P., & Siddharth, N. (2020). Relating by contrasting: A data-efficient framework for multimodal generative models. In International conference on learning representations (ICLR)."},{"key":"10121_CR79","unstructured":"Stafylakis, T., Mo\u0161ner, L., Kakouros, S., Plchot, O., Burget, L., & \u0106ernock\u1ef3, J. (2022). Extracting speaker and emotion information from self-supervised speech models via channel-wise correlations. In IEEE Spoken language technology workshop (SLT)."},{"key":"10121_CR80","unstructured":"Sung, Y.-L., Cho, J., & Bansal, M. (2022). LST: Ladder side-tuning for parameter and memory efficient transfer learning. In Advances in neural information processing systems (NIPS)."},{"key":"10121_CR81","doi-asserted-by":"crossref","unstructured":"Tay, Y., Dehghani, M., Bahri, D., & Metzler, D. (2022). Efficient transformers: A survey. ACM Computing Surveys,6","DOI":"10.1145\/3530811"},{"key":"10121_CR82","doi-asserted-by":"crossref","unstructured":"Tu, Z., Talebi, H., Zhang, H., Yang, F., Milanfar, P., Bovik, A., & Li, Y. (2022). Maxvit: Multi-axis vision transformer. In European conference on computer vision.","DOI":"10.1007\/978-3-031-20053-3_27"},{"key":"10121_CR83","doi-asserted-by":"crossref","unstructured":"Tyagi, S., & Sharma, P. (2020). Taming resource heterogeneity in distributed ml training with dynamic batching. In IEEE international conference on autonomic computing and self-organizing systems (ACSOS).","DOI":"10.1109\/ACSOS49614.2020.00041"},{"key":"10121_CR84","doi-asserted-by":"crossref","unstructured":"Vyas, A., Hsu, W.-N., Auli, M., & Baevski, A. (2022). On-demand compute reduction with stochastic wav2vec 2.0. arXiv preprint arXiv:2204.11934","DOI":"10.21437\/Interspeech.2022-10584"},{"key":"10121_CR85","doi-asserted-by":"crossref","unstructured":"Wang, R., Bai, Q., Ao, J., Zhou, L., Xiong, Z., Wei, Z., Zhang, Y., Ko, T., & Li, H. (2022). Lighthubert: Lightweight and configurable speech representation learning with once-for-all hidden-unit BERT. In Interspeech.","DOI":"10.21437\/Interspeech.2022-10269"},{"key":"10121_CR86","unstructured":"Wang, S., Li, B.Z., Khabsa, M., Fang, H., & Ma, H. (2020). Linformer: Self-attention with linear complexity. arXiv preprint arXiv:2006.04768"},{"key":"10121_CR87","doi-asserted-by":"crossref","unstructured":"Wang, Y., Li, J., Wang, H., Qian, Y., Wang, C., & Wu, Y. (2022). Wav2vec-switch: Contrastive learning from original-noisy speech pairs for robust speech recognition. In IEEE international conference on acoustics, speech and signal processing (ICASSP).","DOI":"10.1109\/ICASSP43922.2022.9746929"},{"key":"10121_CR88","doi-asserted-by":"crossref","unstructured":"Wang, Y., Mohamed, A., Le, D., Liu, C., Xiao, A., Mahadeokar, J., Huang, H., Tjandra, A., Zhang, X., Zhang, F., Fuegen, C., Zweig, G., & Seltzer, M. (2020). Transformer-based acoustic modeling for hybrid speech recognition. In IEEE international conference on acoustics, speech and signal processing (ICASSP).","DOI":"10.1109\/ICASSP40776.2020.9054345"},{"key":"10121_CR89","unstructured":"Wang, S., Nguyen, J., Li, K., & Wu, C.-J. (2023). Read: Recurrent adaptation of large transformers. arXiv preprint arXiv:2305.15348"},{"key":"10121_CR90","doi-asserted-by":"crossref","unstructured":"Wang, A., Singh, A., Michael, J., Hill, F., Levy, O., & Bowman, S. (2018). Glue: A multi-task benchmark and analysis platform for natural language understanding. In Proceedings of the 2018 EMNLP workshop blackbox NLP: Analyzing and interpreting neural networks for NLP.","DOI":"10.18653\/v1\/W18-5446"},{"key":"10121_CR91","doi-asserted-by":"crossref","unstructured":"Wang, S., Zhou, L., Gan, Z., Chen, Y.-C., Fang, Y., Sun, S., Cheng, Y., & Liu, J. (2021). Cluster-former: Clustering-based sparse transformer for question answering. In Proceedings of the annual meeting of the association for computational linguistics and the international joint conference on natural language processing.","DOI":"10.18653\/v1\/2021.findings-acl.346"},{"key":"10121_CR92","unstructured":"Wang, T., Zhu, J.-Y., Torralba, A., & Efros, A.A. (2018). Dataset distillation. arXiv preprint arXiv:1811.10959"},{"key":"10121_CR93","doi-asserted-by":"crossref","unstructured":"Wu, F., Kim, K., Pan, J., Han, K.J., Weinberger, K.Q., & Artzi, Y. (2022). Performance-efficiency trade-offs in unsupervised pre-training for speech recognition. In IEEE international conference on acoustics, speech and signal processing (ICASSP).","DOI":"10.1109\/ICASSP43922.2022.9747432"},{"key":"10121_CR94","unstructured":"Wu, Z., Liu, Z., Lin, J., Lin, Y., & Han, S. (2020). Lite transformer with long-short range attention. In International conference on learning representations (ICLR)."},{"key":"10121_CR95","first-page":"795","volume":"4","author":"C-J Wu","year":"2022","unstructured":"Wu, C.-J., Raghavendra, R., Gupta, U., Acun, B., Ardalani, N., Maeng, K., Chang, G., Aga, F., Huang, J., Bai, C., Gschwind, M., Gupta, A., Ott, M., Melnikov, A., Candido, S., Brooks, D., Chauhan, G., Lee, B., Lee, H.-H., \u2026 Hazelwood, K. (2022). Sustainable AI: Environmental implications, challenges and opportunities. Proceedings of Machine Learning and Systems, 4, 795\u2013813.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"10121_CR96","doi-asserted-by":"crossref","unstructured":"Xie, Q., Luong, M.-T., Hovy, E., & Le, Q.V. (2020). Self-training with noisy student improves imagenet classification. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR42600.2020.01070"},{"key":"10121_CR97","doi-asserted-by":"crossref","unstructured":"Yang, B., Wang, L., Wong, D.F., Chao, L.S., & Tu, Z. (2019). Convolutional self-attention networks. In Proceedings of the conference of the North American chapter of the association for computational linguistics: Human language technologies.","DOI":"10.18653\/v1\/N19-1407"},{"key":"10121_CR98","doi-asserted-by":"crossref","unstructured":"Yang, S.-w., Chi, P.-H., Chuang, Y.-S., Lai, C.-I.J., Lakhotia, K., Lin, Y.Y., Liu, A.T., Shi, J., Chang, X., Lin, G.-T., Huang, T.-H., Tseng, W.-C., Lee, K.-T., Liu, D.-R., Huang, Z., Dong, S., Li, S.-W., Watanabe, S., Mohamed, A., & Lee, H.-Y. (2021). Superb: Speech processing universal performance benchmark. In Interspeech.","DOI":"10.21437\/Interspeech.2021-1775"},{"key":"10121_CR99","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-530","volume-title":"Autoregressive co-training for learning discrete speech representations","author":"S-L Yeh","year":"2022","unstructured":"Yeh, S.-L., & Tang, H. (2022). Autoregressive co-training for learning discrete speech representations. In Interspeech."},{"key":"10121_CR100","unstructured":"Yu, A.W., Dohan, D., Luong, M.-T., Zhao, R., Chen, K., Norouzi, M., & Le, Q.V. (2018). Qanet: Combining local convolution with global self-attention for reading comprehension. In International conference on learning representations (ICLR)."},{"key":"10121_CR101","doi-asserted-by":"crossref","unstructured":"Zaken, E.B., Goldberg, Y., & Ravfogel, S. (2022). Bitfit: Simple parameter-efficient fine-tuning for transformer-based masked language-models. In Proceedings of the 60th annual meeting of the association for computational linguistics.","DOI":"10.18653\/v1\/2022.acl-short.1"},{"key":"10121_CR102","unstructured":"Zhai, S., Talbott, W., Srivastava, N., Huang, C., Goh, H., Zhang, R., & Susskind, J. (2021). An attention free transformer. arXiv preprint arXiv:2105.14103"},{"key":"10121_CR103","unstructured":"Zhang, Q., Chen, M., Bukharin, A., He, P., Cheng, Y., Chen, W., & Zhao, T. (2023). Adaptive budget allocation for parameter-efficient fine-tuning. In International conference on learning representations (ICLR)."},{"key":"10121_CR104","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Chen, G., Yu, D., Yao, K., Khudanpur, S., & Glass, J. (2016). Highway long short-term memory RNNS for distant speech recognition. In IEEE international conference on acoustics, speech and signal processing (ICASSP).","DOI":"10.1109\/ICASSP.2016.7472780"},{"key":"10121_CR105","unstructured":"Zhang, Y., Qin, J., Park, D.S., Han, W., Chiu, C.-C., Pang, R., Le, Q.V., & Wu, Y. (2020). Pushing the limits of semi-supervised learning for automatic speech recognition. arXiv preprint arXiv:2010.10504"},{"key":"10121_CR106","doi-asserted-by":"crossref","unstructured":"Zhang, J.O., Sax, A., Zamir, A., Guibas, L., & Malik, J. (2020). Side-tuning: A baseline for network adaptation via additive side networks. In European conference on computer vision.","DOI":"10.1007\/978-3-030-58580-8_41"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10121-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10772-024-10121-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10121-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,12]],"date-time":"2024-09-12T12:14:00Z","timestamp":1726143240000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10772-024-10121-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,19]]},"references-count":106,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2024,9]]}},"alternative-id":["10121"],"URL":"https:\/\/doi.org\/10.1007\/s10772-024-10121-9","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"value":"1381-2416","type":"print"},{"value":"1572-8110","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,8,19]]},"assertion":[{"value":"22 January 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 June 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 August 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial or non-financial interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}}]}}