{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T05:38:39Z","timestamp":1776922719480,"version":"3.51.2"},"reference-count":48,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100012645","name":"The University of Texas at Dallas","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012645","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Speech Communication"],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1016\/j.specom.2026.103380","type":"journal-article","created":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T07:49:06Z","timestamp":1774338546000},"page":"103380","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Advancing automatic speech recognition using feature fusion with self-supervised learning features: A case study on Fearless Steps Apollo corpus"],"prefix":"10.1016","volume":"180","author":[{"given":"Szu-Jui","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1382-9929","authenticated-orcid":false,"given":"John H.L.","family":"Hansen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.specom.2026.103380_b1","doi-asserted-by":"crossref","first-page":"5145","DOI":"10.21437\/Interspeech.2022-11376","article-title":"Investigation of ensemble features of self-supervised pretrained models for automatic speech recognition","author":"Arunkumar","year":"2022","journal-title":"ISCA Interspeech-2022"},{"key":"10.1016\/j.specom.2026.103380_b2","series-title":"ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"10166","article-title":"What do self-supervised speech and speaker models learn? new findings from a cross model layer-wise analysis","author":"Ashihara","year":"2024"},{"key":"10.1016\/j.specom.2026.103380_b3","first-page":"12449","article-title":"Wav2Vec 2.0: A framework for self-supervised learning of speech representations","volume":"33","author":"Baevski","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.specom.2026.103380_b4","doi-asserted-by":"crossref","first-page":"3533","DOI":"10.21437\/Interspeech.2022-10796","article-title":"Combining spectral and self-supervised features for low resource speech recognition and translation","author":"Berrebbi","year":"2022","journal-title":"ISCA Interspeech-2022"},{"key":"10.1016\/j.specom.2026.103380_b5","series-title":"CHiME5 Workshop, Hyderabad, India","article-title":"Front-end processing for the chime-5 dinner party scenario","volume":"vol. 1","author":"Boeddeker","year":"2018"},{"key":"10.1016\/j.specom.2026.103380_b6","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"17718","article-title":"CAR-transformer: Cross-attention reinforcement transformer for cross-lingual summarization","volume":"vol. 38","author":"Cai","year":"2024"},{"key":"10.1016\/j.specom.2026.103380_b7","first-page":"228","article-title":"An exploration of self-supervised pretrained representations for end-to-end speech recognition","author":"Chang","year":"2021","journal-title":"IEEE ASRU-2021: Autom. Speech Recog. Underst. Work."},{"key":"10.1016\/j.specom.2026.103380_b8","doi-asserted-by":"crossref","unstructured":"Chen, C.-F.R., Fan, Q., Panda, R., 2021. Crossvit: Cross-attention multi-scale vision transformer for image classification. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 357\u2013366.","DOI":"10.1109\/ICCV48922.2021.00041"},{"issue":"6","key":"10.1016\/j.specom.2026.103380_b9","doi-asserted-by":"crossref","first-page":"1505","DOI":"10.1109\/JSTSP.2022.3188113","article-title":"WavLM: Large-scale self-supervised pre-training for full stack speech processing","volume":"16","author":"Chen","year":"2022","journal-title":"IEEE J. Sel. Top. Signal Process."},{"key":"10.1016\/j.specom.2026.103380_b10","first-page":"289","article-title":"Scenario Aware Speech Recognition: Advancements for Apollo Fearless Steps & CHiME-4 Corpora","author":"Chen","year":"2021","journal-title":"IEEE ASRU-2021: Autom. Speech Recog. Underst. Work."},{"key":"10.1016\/j.specom.2026.103380_b11","doi-asserted-by":"crossref","DOI":"10.21437\/Interspeech.2022-10917","article-title":"FeaRLESS: Feature Refinement Loss for Ensembling Self-Supervised Learning Features in Robust End-to-end Speech Recognition","author":"Chen","year":"2022","journal-title":"ISCA Interspeech-2022"},{"key":"10.1016\/j.specom.2026.103380_b12","first-page":"6170","article-title":"XLM-e: Cross-lingual language model pre-training via ELECTRA","author":"Chi","year":"2022","journal-title":"Annual Meet. Assoc. Comp. Ling."},{"key":"10.1016\/j.specom.2026.103380_b13","doi-asserted-by":"crossref","unstructured":"Chiu, S.-C., Wu, C.-H., Hsieh, J.-K., Tsao, Y., Wang, H.-M., 2024. Learnable Layer Selection and Model Fusion for Speech Self-Supervised Learning Models. In: Proc. Interspeech 2024. pp. 3914\u20133918.","DOI":"10.21437\/Interspeech.2024-849"},{"key":"10.1016\/j.specom.2026.103380_b14","doi-asserted-by":"crossref","first-page":"1509","DOI":"10.21437\/Interspeech.2021-1280","article-title":"Exploring Wav2Vec 2.0 on speaker verification and language identification","author":"Fan","year":"2021","journal-title":"ISCA Interspeech-2021"},{"key":"10.1016\/j.specom.2026.103380_b15","series-title":"NIST SCTK Toolkit","author":"Fiscus","year":"2018"},{"key":"10.1016\/j.specom.2026.103380_b16","series-title":"International Conference on Acoustics, Speech, and Signal Processing","first-page":"532","article-title":"Some statistical issues in the comparison of speech recognition algorithms","author":"Gillick","year":"1989"},{"key":"10.1016\/j.specom.2026.103380_b17","doi-asserted-by":"crossref","first-page":"2612","DOI":"10.21437\/Interspeech.2020-2822","article-title":"\u201cThis is Houston. Say again, please.\u201d The Behavox system for the Apollo-11 Fearless Steps Challenge (Phase II)","author":"Gorin","year":"2020","journal-title":"ISCA Interspeech-2020"},{"key":"10.1016\/j.specom.2026.103380_b18","doi-asserted-by":"crossref","first-page":"5036","DOI":"10.21437\/Interspeech.2020-3015","article-title":"Conformer: Convolution-augmented Transformer for Speech Recognition","author":"Gulati","year":"2020","journal-title":"Proc. Interspeech 2020"},{"key":"10.1016\/j.specom.2026.103380_b19","series-title":"ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"12816","article-title":"Fearless steps apollo: Team communications based community resource development for science, technology, education, and historical preservation","author":"Hansen","year":"2024"},{"key":"10.1016\/j.specom.2026.103380_b20","doi-asserted-by":"crossref","first-page":"1851","DOI":"10.21437\/Interspeech.2019-2301","article-title":"The 2019 Inaugural Fearless Steps Challenge: A Giant Leap for Naturalistic Audio","author":"Hansen","year":"2019","journal-title":"ISCA Interspeech-2019"},{"key":"10.1016\/j.specom.2026.103380_b21","doi-asserted-by":"crossref","first-page":"2758","DOI":"10.21437\/Interspeech.2018-1942","article-title":"Fearless steps: Apollo-11 corpus advancements for speech technologies from earth to the moon","author":"Hansen","year":"2018","journal-title":"ISCA Interspeech-2018"},{"key":"10.1016\/j.specom.2026.103380_b22","series-title":"Gaussian error linear units (gelus)","author":"Hendrycks","year":"2016"},{"key":"10.1016\/j.specom.2026.103380_b23","doi-asserted-by":"crossref","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","article-title":"Hubert: Self-supervised speech representation learning by masked prediction of hidden units","volume":"29","author":"Hsu","year":"2021","journal-title":"IEEE\/ACM Trans. Audio, Speech, Lang. Proc."},{"key":"10.1016\/j.specom.2026.103380_b24","doi-asserted-by":"crossref","first-page":"2617","DOI":"10.21437\/Interspeech.2020-3054","article-title":"FEARLESS STEPS Challenge (FS-2): Supervised Learning with Massive Naturalistic Apollo Data","author":"Joglekar","year":"2020","journal-title":"ISCA Interspeech-2020"},{"key":"10.1016\/j.specom.2026.103380_b25","doi-asserted-by":"crossref","first-page":"986","DOI":"10.21437\/Interspeech.2021-2011","article-title":"Fearless Steps Challenge Phase-3 (FSC P3): Advancing SLT for Unseen Channel and Mission Data Across NASA Apollo Audio","author":"Joglekar","year":"2021","journal-title":"ISCA Interspeech-2021"},{"key":"10.1016\/j.specom.2026.103380_b26","doi-asserted-by":"crossref","unstructured":"Kim, K., Wu, F., Peng, Y., Pan, J., Sridhar, P., Han, K.J., Watanabe, S., 2023. E-Branchformer: Branchformer with Enhanced merging for speech recognition. In: SLT-23: IEEE Spoken Lang. Tech. Workshop. pp. 84\u201391.","DOI":"10.1109\/SLT54892.2023.10022656"},{"key":"10.1016\/j.specom.2026.103380_b27","unstructured":"Kingma, D.P., Ba, J., 2014. Adam: A method for stochastic optimization. In: International Conference on Learning Representations."},{"key":"10.1016\/j.specom.2026.103380_b28","series-title":"2022 IEEE International Conference on Multimedia and Expo","first-page":"1","article-title":"Cat: Cross attention in vision transformer","author":"Lin","year":"2022"},{"key":"10.1016\/j.specom.2026.103380_b29","series-title":"Decoupled weight decay regularization","author":"Loshchilov","year":"2017"},{"issue":"6","key":"10.1016\/j.specom.2026.103380_b30","doi-asserted-by":"crossref","first-page":"1179","DOI":"10.1109\/JSTSP.2022.3207050","article-title":"Self-supervised speech representation learning: A review","volume":"16","author":"Mohamed","year":"2022","journal-title":"IEEE J. Sel. Top. Signal Process."},{"key":"10.1016\/j.specom.2026.103380_b31","doi-asserted-by":"crossref","DOI":"10.21437\/Interspeech.2020-1835","article-title":"Investigating self-supervised pre-training for end-to-end speech translation","author":"Nguyen","year":"2020","journal-title":"ISCA Interspeech-2020"},{"key":"10.1016\/j.specom.2026.103380_b32","unstructured":"Oord, A.v.d., Li, Y., Vinyals, O., 2018. Representation learning with contrastive predictive coding. In: Proc. of NIPS."},{"key":"10.1016\/j.specom.2026.103380_b33","first-page":"5206","article-title":"Librispeech: an asr corpus based on public domain audio books","author":"Panayotov","year":"2015"},{"key":"10.1016\/j.specom.2026.103380_b34","first-page":"2613","article-title":"Specaugment: A simple data augmentation method for automatic speech recognition","author":"Park","year":"2019","journal-title":"Proc. Annu. Conf. Int. Speech Commun. Assoc."},{"key":"10.1016\/j.specom.2026.103380_b35","series-title":"ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"Comparative layer-wise analysis of self-supervised speech models","author":"Pasad","year":"2023"},{"key":"10.1016\/j.specom.2026.103380_b36","doi-asserted-by":"crossref","first-page":"3400","DOI":"10.21437\/Interspeech.2021-703","article-title":"Emotion recognition from speech using Wav2Vec 2.0 embeddings","author":"Pepino","year":"2021","journal-title":"ISCA Interspeech-2021"},{"key":"10.1016\/j.specom.2026.103380_b37","first-page":"28492","article-title":"Robust speech recognition via large-scale weak supervision","author":"Radford","year":"2023","journal-title":"Intern. Conf. Mach. Learn."},{"key":"10.1016\/j.specom.2026.103380_b38","doi-asserted-by":"crossref","unstructured":"Srivastava, T., Shi, J., Chen, W., Watanabe, S., 2024. EFFUSE: Efficient Self-Supervised Feature Fusion for E2E ASR in Low Resource and Multilingual Scenarios. In: Proc. Interspeech 2024. pp. 3989\u20133993.","DOI":"10.21437\/Interspeech.2024-2199"},{"key":"10.1016\/j.specom.2026.103380_b39","doi-asserted-by":"crossref","DOI":"10.21437\/Interspeech.2021-211","article-title":"On the limit of English conversational speech recognition","author":"T\u00fcske","year":"2021","journal-title":"ISCA Interspeech-2021"},{"key":"10.1016\/j.specom.2026.103380_b40","first-page":"5998","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Adv. Neural Info. Proc. Syst."},{"key":"10.1016\/j.specom.2026.103380_b41","series-title":"2024 IEEE Spoken Language Technology Workshop","first-page":"247","article-title":"Fusion of discrete representations and self-augmented representations for multilingual automatic speech recognition","author":"Wang","year":"2024"},{"key":"10.1016\/j.specom.2026.103380_b42","doi-asserted-by":"crossref","first-page":"2207","DOI":"10.21437\/Interspeech.2018-1456","article-title":"Espnet: End-to-end speech processing toolkit","author":"Watanabe","year":"2018","journal-title":"Proc. Interspeech 2018"},{"issue":"8","key":"10.1016\/j.specom.2026.103380_b43","doi-asserted-by":"crossref","first-page":"1240","DOI":"10.1109\/JSTSP.2017.2763455","article-title":"Hybrid CTC\/attention architecture for End-to-End speech recognition","volume":"11","author":"Watanabe","year":"2017","journal-title":"IEEE J. Sel. Top. Signal Process."},{"key":"10.1016\/j.specom.2026.103380_b44","doi-asserted-by":"crossref","unstructured":"Watanabe, S., Mandel, M., Barker, J., Vincent, E., Arora, A., Chang, X., Khudanpur, S., Manohar, V., Povey, D., Raj, D., et al., 2020. CHiME-6 challenge: Tackling multispeaker speech recognition for unsegmented recordings. In: Workshop on Speech Processing in Everyday Environments (CHiME 2020). pp. 1\u20137.","DOI":"10.21437\/CHiME.2020-1"},{"key":"10.1016\/j.specom.2026.103380_b45","doi-asserted-by":"crossref","first-page":"1491","DOI":"10.21437\/Interspeech.2020-3094","article-title":"Self-supervised representations improve end-to-end speech translation","author":"Wu","year":"2020","journal-title":"ISCA Interspeech-2020"},{"key":"10.1016\/j.specom.2026.103380_b46","doi-asserted-by":"crossref","unstructured":"Xiong, W., Wu, L., Alleva, F., Droppo, J., Huang, X., Stolcke, A., 2018. The Microsoft 2017 conversational speech recognition system. In: IEEE ICASSP-2018: Inter. Conf. on Acoustics, Speech and Signal Proc.. pp. 5934\u20135938.","DOI":"10.1109\/ICASSP.2018.8461870"},{"key":"10.1016\/j.specom.2026.103380_b47","doi-asserted-by":"crossref","first-page":"1194","DOI":"10.21437\/Interspeech.2021-1775","article-title":"Superb: Speech processing universal performance benchmark","author":"Yang","year":"2021","journal-title":"ISCA Interspeech-2021"},{"key":"10.1016\/j.specom.2026.103380_b48","series-title":"Applying Wav2Vec 2.0 to speech recognition in various low-resource languages","author":"Yi","year":"2020"}],"container-title":["Speech Communication"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167639326000282?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167639326000282?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T04:46:29Z","timestamp":1776919589000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0167639326000282"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5]]},"references-count":48,"alternative-id":["S0167639326000282"],"URL":"https:\/\/doi.org\/10.1016\/j.specom.2026.103380","relation":{},"ISSN":["0167-6393"],"issn-type":[{"value":"0167-6393","type":"print"}],"subject":[],"published":{"date-parts":[[2026,5]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Advancing automatic speech recognition using feature fusion with self-supervised learning features: A case study on Fearless Steps Apollo corpus","name":"articletitle","label":"Article Title"},{"value":"Speech Communication","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.specom.2026.103380","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"103380"}}