{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T05:39:17Z","timestamp":1776922757423,"version":"3.51.2"},"reference-count":90,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62301096"],"award-info":[{"award-number":["62301096"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100005230","name":"Natural Science Foundation of Chongqing Municipality","doi-asserted-by":"publisher","award":["CSTB2023NSCQ-MSX0659"],"award-info":[{"award-number":["CSTB2023NSCQ-MSX0659"]}],"id":[{"id":"10.13039\/501100005230","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012669","name":"Natural Science Foundation Project of Chongqing","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012669","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002855","name":"Ministry of Science and Technology of the People's Republic of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100002855","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2024QY2630"],"award-info":[{"award-number":["2024QY2630"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Speech Communication"],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1016\/j.specom.2026.103395","type":"journal-article","created":{"date-parts":[[2026,3,31]],"date-time":"2026-03-31T06:46:50Z","timestamp":1774939610000},"page":"103395","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["MaTSE: A hybrid Mamba-Transformer model for monaural Speech Enhancement"],"prefix":"10.1016","volume":"180","author":[{"given":"Hao","family":"Zhou","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yi","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yu","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yin","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.specom.2026.103395_b1","doi-asserted-by":"crossref","first-page":"2477","DOI":"10.1109\/TASLP.2024.3393718","article-title":"CMGAN: Conformer-based metric-GAN for monaural speech enhancement","volume":"32","author":"Abdulatif","year":"2024","journal-title":"IEEE\/ACM Trans. Audio, Speech, Lang. Process."},{"key":"10.1016\/j.specom.2026.103395_b2","series-title":"ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"SepMamba: State-space models for speaker separation using mamba","author":"Avenstrup","year":"2025"},{"key":"10.1016\/j.specom.2026.103395_b3","unstructured":"Botinhao, C.V., Wang, X., Takaki, S., Yamagishi, J., 2016. Investigating RNN-based speech enhancement methods for noise-robust text-to-speech. In: 9th ISCA Speech Synthesis Workshop. pp. 159\u2013165."},{"key":"10.1016\/j.specom.2026.103395_b4","series-title":"2024 IEEE Spoken Language Technology Workshop","first-page":"302","article-title":"An investigation of incorporating mamba for speech enhancement","author":"Chao","year":"2024"},{"key":"10.1016\/j.specom.2026.103395_b5","doi-asserted-by":"crossref","unstructured":"Chen, J., Mao, Q., Liu, D., 2020. Dual-Path Transformer Network: Direct Context-Aware Modeling for End-to-End Monaural Speech Separation. In: Proc. Interspeech 2020. pp. 2642\u20132646.","DOI":"10.21437\/Interspeech.2020-2205"},{"key":"10.1016\/j.specom.2026.103395_b6","series-title":"ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"Inter-subnet: Speech enhancement with subband interaction","author":"Chen","year":"2023"},{"key":"10.1016\/j.specom.2026.103395_b7","doi-asserted-by":"crossref","unstructured":"Chen, J., Rao, W., Wang, Z., Wu, Z., Wang, Y., Yu, T., Shang, S., Meng, H., 2022b. Speech Enhancement with Fullband-Subband Cross-Attention Network. In: Proc. Interspeech 2022. pp. 976\u2013980.","DOI":"10.21437\/Interspeech.2022-10257"},{"key":"10.1016\/j.specom.2026.103395_b8","series-title":"ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"7857","article-title":"Fullsubnet+: Channel attention fullsubnet with complex spectrograms for speech enhancement","author":"Chen","year":"2022"},{"key":"10.1016\/j.specom.2026.103395_b9","article-title":"Selective state space model for monaural speech enhancement","author":"Chen","year":"2025","journal-title":"IEEE Trans. Consum. Electron."},{"key":"10.1016\/j.specom.2026.103395_b10","series-title":"ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"6857","article-title":"DPT-FSNet: Dual-path transformer based full-band and sub-band fusion network for speech enhancement","author":"Dang","year":"2022"},{"key":"10.1016\/j.specom.2026.103395_b11","series-title":"THLNet: two-stage heterogeneous lightweight network for monaural speech enhancement","author":"Dang","year":"2023"},{"key":"10.1016\/j.specom.2026.103395_b12","series-title":"ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"766","article-title":"Spiking structured state space model for monaural speech enhancement","author":"Du","year":"2024"},{"key":"10.1016\/j.specom.2026.103395_b13","series-title":"2018 26th European Signal Processing Conference","first-page":"390","article-title":"Speech dereverberation using fully convolutional networks","author":"Ernst","year":"2018"},{"key":"10.1016\/j.specom.2026.103395_b14","doi-asserted-by":"crossref","DOI":"10.1109\/LSP.2024.3449221","article-title":"The effect of training dataset size on discriminative and diffusion-based speech enhancement systems","author":"Gonzalez","year":"2024","journal-title":"IEEE Signal Process. Lett."},{"key":"10.1016\/j.specom.2026.103395_b15","series-title":"Mamba: Linear-time sequence modeling with selective state spaces","author":"Gu","year":"2023"},{"key":"10.1016\/j.specom.2026.103395_b16","first-page":"35971","article-title":"On the parameterization and initialization of diagonal state space models","volume":"35","author":"Gu","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.specom.2026.103395_b17","series-title":"International Conference on Learning Representations","article-title":"Efficiently modeling long sequences with structured state spaces","author":"Gu","year":"2022"},{"key":"10.1016\/j.specom.2026.103395_b18","doi-asserted-by":"crossref","unstructured":"Gulati, A., Qin, J., Chiu, C.-C., Parmar, N., Zhang, Y., Yu, J., Han, W., Wang, S., Zhang, Z., Wu, Y., et al., 2020. Conformer: Convolution-augmented Transformer for Speech Recognition. In: Proc. Interspeech 2020. pp. 5036\u20135040.","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"10.1016\/j.specom.2026.103395_b19","series-title":"ICASSP 2021-2021 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"6633","article-title":"Fullsubnet: A full-band and sub-band fusion model for real-time single-channel speech enhancement","author":"Hao","year":"2021"},{"key":"10.1016\/j.specom.2026.103395_b20","doi-asserted-by":"crossref","unstructured":"Hatamizadeh, A., Kautz, J., 2025. Mambavision: A hybrid mamba-transformer vision backbone. In: Proceedings of the Computer Vision and Pattern Recognition Conference. pp. 25261\u201325270.","DOI":"10.1109\/CVPR52734.2025.02352"},{"key":"10.1016\/j.specom.2026.103395_b21","article-title":"100 Nonspeech environmental sounds","author":"Hu","year":"2004","journal-title":"Ohio State Univ. Dep. Comput. Sci. Eng."},{"key":"10.1016\/j.specom.2026.103395_b22","series-title":"2024 4th International Symposium on Artificial Intelligence and Intelligent Manufacturing","first-page":"708","article-title":"Mcodec: A codec architecture integrated mamba for speech enhancement","author":"Hu","year":"2024"},{"key":"10.1016\/j.specom.2026.103395_b23","series-title":"Bidirectional LSTM-CRF models for sequence tagging","author":"Huang","year":"2015"},{"issue":"11","key":"10.1016\/j.specom.2026.103395_b24","doi-asserted-by":"crossref","first-page":"2009","DOI":"10.1109\/TASLP.2016.2585878","article-title":"An algorithm for predicting the intelligibility of speech masked by modulated noise maskers","volume":"24","author":"Jensen","year":"2016","journal-title":"IEEE\/ACM Trans. Audio, Speech, Lang. Process."},{"issue":"2","key":"10.1016\/j.specom.2026.103395_b25","doi-asserted-by":"crossref","first-page":"125","DOI":"10.1109\/TCE.2020.2986003","article-title":"Speech enhancement parameter adjustment to maximize accuracy of automatic speech recognition","volume":"66","author":"Kawase","year":"2020","journal-title":"IEEE Trans. Consum. Electron."},{"key":"10.1016\/j.specom.2026.103395_b26","doi-asserted-by":"crossref","unstructured":"Kim, S.-H., Kim, T.-G., Chun, C.-J., 2025. Mamba-based Hybrid Model for Speech Enhancement. In: Proceedings of the Interspeech, Rotterdam, The Netherlands. pp. 17\u201321.","DOI":"10.21437\/Interspeech.2025-1476"},{"key":"10.1016\/j.specom.2026.103395_b27","series-title":"Interspeech","first-page":"2736","article-title":"SE-conformer: Time-domain speech enhancement using conformer.","author":"Kim","year":"2021"},{"key":"10.1016\/j.specom.2026.103395_b28","series-title":"Adam: A method for stochastic optimization","author":"Kingma","year":"2014"},{"issue":"1","key":"10.1016\/j.specom.2026.103395_b29","doi-asserted-by":"crossref","first-page":"7","DOI":"10.1186\/s13634-016-0306-6","article-title":"A summary of the REVERB challenge: state-of-the-art and remaining challenges in reverberant speech processing research","volume":"2016","author":"Kinoshita","year":"2016","journal-title":"EURASIP J. Adv. Signal Process."},{"key":"10.1016\/j.specom.2026.103395_b30","series-title":"International Conference on Latent Variable Analysis and Signal Separation","first-page":"243","article-title":"Sparsity and cosparsity for audio declipping: a flexible non-convex approach","author":"Kiti\u0107","year":"2015"},{"key":"10.1016\/j.specom.2026.103395_b31","series-title":"Description and discussion on DCASE2020 challenge task2: Unsupervised anomalous sound detection for machine condition monitoring","author":"Koizumi","year":"2020"},{"key":"10.1016\/j.specom.2026.103395_b32","series-title":"International Conference on Machine Learning","first-page":"3519","article-title":"Similarity of neural network representations revisited","author":"Kornblith","year":"2019"},{"key":"10.1016\/j.specom.2026.103395_b33","doi-asserted-by":"crossref","unstructured":"Kothapally, V., Xia, W., Ghorbani, S., Hansen, J.H., Xue, W., Huang, J., 2020. SkipConvNet: Skip Convolutional Neural Network for Speech Dereverberation Using Optimally Smoothed Spectral Mapping. In: Proc. Interspeech 2020. pp. 3935\u20133939.","DOI":"10.21437\/Interspeech.2020-2048"},{"key":"10.1016\/j.specom.2026.103395_b34","doi-asserted-by":"crossref","unstructured":"Le, X., Chen, H., Chen, K., Lu, J., 2021. DPCRN: Dual-Path Convolution Recurrent Network for Single Channel Speech Enhancement. In: Proc. Interspeech 2021. pp. 2811\u20132815.","DOI":"10.21437\/Interspeech.2021-296"},{"key":"10.1016\/j.specom.2026.103395_b35","doi-asserted-by":"crossref","DOI":"10.1109\/TASLP.2024.3488564","article-title":"DeFTAN-II: Efficient multichannel speech enhancement with subgroup processing","author":"Lee","year":"2024","journal-title":"IEEE\/ACM Trans. Audio, Speech, Lang. Process."},{"key":"10.1016\/j.specom.2026.103395_b36","series-title":"Spmamba: State-space model is all you need in speech separation","author":"Li","year":"2024"},{"key":"10.1016\/j.specom.2026.103395_b37","doi-asserted-by":"crossref","first-page":"1829","DOI":"10.1109\/TASLP.2021.3079813","article-title":"Two heads are better than one: A two-stage complex spectral mapping approach for monaural speech enhancement","volume":"29","author":"Li","year":"2021","journal-title":"IEEE\/ACM Trans. Audio, Speech, Lang. Process."},{"key":"10.1016\/j.specom.2026.103395_b38","doi-asserted-by":"crossref","first-page":"1511","DOI":"10.1109\/TASLP.2023.3265839","article-title":"U-shaped transformer with frequency-band aware attention for speech enhancement","volume":"31","author":"Li","year":"2023","journal-title":"IEEE\/ACM Trans. Audio, Speech, Lang. Process."},{"key":"10.1016\/j.specom.2026.103395_b39","series-title":"ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"10511","article-title":"Lightweight multi-axial transformer with frequency prompt for single channel speech enhancement","author":"Liang","year":"2024"},{"key":"10.1016\/j.specom.2026.103395_b40","series-title":"ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"Primek-net: Multi-scale spectral learning via group prime-kernel convolutional neural networks for single channel speech enhancement","author":"Lin","year":"2025"},{"key":"10.1016\/j.specom.2026.103395_b41","doi-asserted-by":"crossref","unstructured":"Liu, L., Guan, H., Ma, J., Dai, W., Wang, G., Ding, S., 2023. A Mask Free Neural Network for Monaural Speech Enhancement. In: Proc. Interspeech 2023. pp. 2468\u20132472.","DOI":"10.21437\/Interspeech.2023-339"},{"key":"10.1016\/j.specom.2026.103395_b42","article-title":"Speech conv-mamba: Selective structured state space model with temporal dilated convolution for efficient speech separation","author":"Liu","year":"2025","journal-title":"IEEE Signal Process. Lett."},{"key":"10.1016\/j.specom.2026.103395_b43","doi-asserted-by":"crossref","unstructured":"Lu, Y.-X., Ai, Y., Ling, Z.-H., 2023. MP-SENet: A Speech Enhancement Model with Parallel Denoising of Magnitude and Phase Spectra. In: Proc. Interspeech 2023. pp. 3834\u20133838.","DOI":"10.21437\/Interspeech.2023-1441"},{"key":"10.1016\/j.specom.2026.103395_b44","article-title":"Explicit estimation of magnitude and phase spectra in parallel for high-quality speech enhancement","author":"Lu","year":"2025","journal-title":"Neural Netw."},{"key":"10.1016\/j.specom.2026.103395_b45","series-title":"ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"46","article-title":"Dual-path rnn: efficient long sequence modeling for time-domain single-channel speech separation","author":"Luo","year":"2020"},{"issue":"8","key":"10.1016\/j.specom.2026.103395_b46","doi-asserted-by":"crossref","first-page":"1256","DOI":"10.1109\/TASLP.2019.2915167","article-title":"Conv-tasnet: Surpassing ideal time\u2013frequency magnitude masking for speech separation","volume":"27","author":"Luo","year":"2019","journal-title":"IEEE\/ACM Trans. Audio, Speech, Lang. Process."},{"key":"10.1016\/j.specom.2026.103395_b47","series-title":"2024 International Conference on Asian Language Processing","first-page":"411","article-title":"MambaGAN: Mamba based metric GAN for monaural speech enhancement","author":"Luo","year":"2024"},{"key":"10.1016\/j.specom.2026.103395_b48","series-title":"2024 IEEE Spoken Language Technology Workshop","first-page":"1","article-title":"Mamba-based decoder-only approach with bidirectional speech modeling for speech recognition","author":"Masuyama","year":"2024"},{"key":"10.1016\/j.specom.2026.103395_b49","doi-asserted-by":"crossref","first-page":"574","DOI":"10.1016\/j.csl.2016.11.003","article-title":"Speech enhancement for robust automatic speech recognition: Evaluation using a baseline system and instrumental measures","volume":"46","author":"Moore","year":"2017","journal-title":"Comput. Speech Lang."},{"key":"10.1016\/j.specom.2026.103395_b50","series-title":"ICASSP 2021-2021 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"7153","article-title":"Cascaded time+ time-frequency unet for speech enhancement: Jointly addressing clipping, codec distortions, and gaps","author":"Nair","year":"2021"},{"key":"10.1016\/j.specom.2026.103395_b51","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2025.130798","article-title":"Cross-architecture knowledge distillation for speech enhancement: From CMGAN to unet","author":"Nguyen","year":"2025","journal-title":"Neurocomputing"},{"issue":"Suppl 3","key":"10.1016\/j.specom.2026.103395_b52","doi-asserted-by":"crossref","first-page":"3651","DOI":"10.1007\/s10462-023-10612-2","article-title":"Deep neural network techniques for monaural speech enhancement and separation: state of the art analysis","volume":"56","author":"Ochieng","year":"2023","journal-title":"Artif. Intell. Rev."},{"key":"10.1016\/j.specom.2026.103395_b53","series-title":"2015 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"5206","article-title":"Librispeech: an asr corpus based on public domain audio books","author":"Panayotov","year":"2015"},{"key":"10.1016\/j.specom.2026.103395_b54","series-title":"Swish: a self-gated activation function","first-page":"5","author":"Ramachandran","year":"2017"},{"key":"10.1016\/j.specom.2026.103395_b55","series-title":"The interspeech 2020 deep noise suppression challenge: Datasets, subjective testing framework, and challenge results","author":"Reddy","year":"2020"},{"key":"10.1016\/j.specom.2026.103395_b56","series-title":"Deep speech enhancement for reverberated and noisy signals using wide residual networks","author":"Ribas","year":"2019"},{"key":"10.1016\/j.specom.2026.103395_b57","series-title":"2024 18th International Workshop on Acoustic Signal Enhancement","first-page":"205","article-title":"TF-locoformer: Transformer with local modeling by convolution for speech separation and enhancement","author":"Saijo","year":"2024"},{"key":"10.1016\/j.specom.2026.103395_b58","doi-asserted-by":"crossref","unstructured":"Salamon, J., Jacoby, C., Bello, J.P., 2014. A dataset and taxonomy for urban sound research. In: Proceedings of the 22nd ACM International Conference on Multimedia. pp. 1041\u20131044.","DOI":"10.1145\/2647868.2655045"},{"key":"10.1016\/j.specom.2026.103395_b59","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2025.113452","article-title":"CTSE-net: Resource-efficient convolutional and TF-transformer network for speech enhancement","volume":"317","author":"Saleem","year":"2025","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.specom.2026.103395_b60","doi-asserted-by":"crossref","DOI":"10.1016\/j.dsp.2024.104408","article-title":"Time domain speech enhancement with CNN and time-attention transformer","volume":"147","author":"Saleem","year":"2024","journal-title":"Digit. Signal Process."},{"key":"10.1016\/j.specom.2026.103395_b61","first-page":"52215","article-title":"Separate and reconstruct: Asymmetric encoder-decoder for speech separation","volume":"37","author":"Shin","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.specom.2026.103395_b62","series-title":"Musan: A music, speech, and noise corpus","author":"Snyder","year":"2015"},{"key":"10.1016\/j.specom.2026.103395_b63","first-page":"1988","article-title":"Description of the RSG-10 noise database","volume":"3","author":"Steeneken","year":"1988","journal-title":"Rep. IZF"},{"key":"10.1016\/j.specom.2026.103395_b64","series-title":"ICASSP 2021-2021 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"21","article-title":"Attention is all you need in speech separation","author":"Subakan","year":"2021"},{"issue":"4","key":"10.1016\/j.specom.2026.103395_b65","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3699757","article-title":"Tramba: A hybrid transformer and mamba architecture for practical audio and bone conduction speech super resolution and enhancement on mobile and wearable platforms","volume":"8","author":"Sui","year":"2024","journal-title":"Proc. ACM Interact. Mob. Wearable Ubiquitous Technol."},{"key":"10.1016\/j.specom.2026.103395_b66","doi-asserted-by":"crossref","unstructured":"Sui, Y., Zhao, M., Xia, J., Jiang, X., Xia, S., 2024b. TraMSR: Transformer and Mamba based Practical Speech Super-Resolution for Mobile Wearables. In: Proceedings of the 30th Annual International Conference on Mobile Computing and Networking. pp. 1686\u20131688.","DOI":"10.1145\/3636534.3697460"},{"key":"10.1016\/j.specom.2026.103395_b67","doi-asserted-by":"crossref","first-page":"1457","DOI":"10.1109\/TASLP.2024.3362691","article-title":"Dual-branch modeling based on state-space model for speech enhancement","volume":"32","author":"Sun","year":"2024","journal-title":"IEEE\/ACM Trans. Audio, Speech, Lang. Process."},{"issue":"3","key":"10.1016\/j.specom.2026.103395_b68","doi-asserted-by":"crossref","first-page":"247","DOI":"10.1016\/0167-6393(93)90095-3","article-title":"Assessment for automatic speech recognition: Ii. NOISEX-92: A database and an experiment to study the effect of additive noise on speech recognition systems","volume":"12","author":"Varga","year":"1993","journal-title":"Speech Commun."},{"key":"10.1016\/j.specom.2026.103395_b69","doi-asserted-by":"crossref","DOI":"10.1109\/TCE.2025.3598007","article-title":"Lightweight adaptive deep learning for efficient real-time speech enhancement on edge devices","author":"Wahab","year":"2025","journal-title":"IEEE Trans. Consum. Electron."},{"key":"10.1016\/j.specom.2026.103395_b70","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2024.129150","article-title":"MA-net: Resource-efficient multi-attentional network for end-to-end speech enhancement","volume":"619","author":"Wahab","year":"2025","journal-title":"Neurocomputing"},{"key":"10.1016\/j.specom.2026.103395_b71","doi-asserted-by":"crossref","unstructured":"Wang, J., 2023. Efficient Encoder-Decoder and Dual-Path Conformer for Comprehensive Feature Learning in Speech Enhancement. In: Proc. Interspeech 2023. pp. 2853\u20132857.","DOI":"10.21437\/Interspeech.2023-815"},{"key":"10.1016\/j.specom.2026.103395_b72","series-title":"ICASSP 2021-2021 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"7098","article-title":"TSTNN: Two-stage transformer based neural network for speech enhancement in the time domain","author":"Wang","year":"2021"},{"issue":"3","key":"10.1016\/j.specom.2026.103395_b73","doi-asserted-by":"crossref","DOI":"10.1049\/sil2.12182","article-title":"Multi-stage attention network for monaural speech enhancement","volume":"17","author":"Wang","year":"2023","journal-title":"IET Signal Process."},{"key":"10.1016\/j.specom.2026.103395_b74","series-title":"Graph-mamba: Towards long-range graph sequence modeling with selective state spaces","author":"Wang","year":"2024"},{"issue":"1","key":"10.1016\/j.specom.2026.103395_b75","doi-asserted-by":"crossref","first-page":"21","DOI":"10.1186\/s13636-023-00283-w","article-title":"Time-domain adaptive attention network for single-channel speech separation","volume":"2023","author":"Wang","year":"2023","journal-title":"EURASIP J. Audio, Speech, Music. Process."},{"key":"10.1016\/j.specom.2026.103395_b76","doi-asserted-by":"crossref","unstructured":"Wichern, G., Antognini, J., Flynn, M., Zhu, L.R., McQuinn, E., Crow, D., Manilow, E., Roux, J.L., 2019. WHAM!: Extending Speech Separation to Noisy Environments. In: Proc. Interspeech 2019. pp. 1368\u20131372.","DOI":"10.21437\/Interspeech.2019-2821"},{"key":"10.1016\/j.specom.2026.103395_b77","series-title":"Proc. REVERB Challenge Workshop","first-page":"o2","article-title":"The NTU-ADSC systems for reverberation challenge 2014","author":"Xiao","year":"2014"},{"key":"10.1016\/j.specom.2026.103395_b78","series-title":"European Conference on Computer Vision","first-page":"34","article-title":"PackMamba: Efficient processing of variable-length sequences in mamba training","author":"Xu","year":"2025"},{"key":"10.1016\/j.specom.2026.103395_b79","doi-asserted-by":"crossref","DOI":"10.1016\/j.specom.2025.103248","article-title":"Dual-path and interactive UNET for speech enhancement with multi-order fractional features","author":"Xu","year":"2025","journal-title":"Speech Commun."},{"key":"10.1016\/j.specom.2026.103395_b80","series-title":"ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"801","article-title":"DDD: A perceptually superior low-response-time DNN-based declipper","author":"Yi","year":"2024"},{"key":"10.1016\/j.specom.2026.103395_b81","series-title":"ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"7847","article-title":"Dual-branch attention-in-attention transformer for single-channel speech enhancement","author":"Yu","year":"2022"},{"key":"10.1016\/j.specom.2026.103395_b82","series-title":"European Conference on Computer Vision","first-page":"265","article-title":"Motion mamba: Efficient and long sequence motion generation","author":"Zhang","year":"2024"},{"key":"10.1016\/j.specom.2026.103395_b83","series-title":"ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"Rethinking mamba in speech processing by self-supervised models","author":"Zhang","year":"2025"},{"key":"10.1016\/j.specom.2026.103395_b84","doi-asserted-by":"crossref","first-page":"462","DOI":"10.1109\/TASLP.2022.3225649","article-title":"A time-frequency attention module for neural speech enhancement","volume":"31","author":"Zhang","year":"2022","journal-title":"IEEE\/ACM Trans. Audio, Speech, Lang. Process."},{"key":"10.1016\/j.specom.2026.103395_b85","article-title":"Root mean square layer normalization","volume":"32","author":"Zhang","year":"2019","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.specom.2026.103395_b86","series-title":"ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"Hybrid feature global attention network for noisy-reverberant speech enhancement","author":"Zhang","year":"2025"},{"key":"10.1016\/j.specom.2026.103395_b87","article-title":"Mamba in speech: Towards an alternative to self-attention","author":"Zhang","year":"2025","journal-title":"IEEE Trans. Audio, Speech Lang. Process."},{"key":"10.1016\/j.specom.2026.103395_b88","series-title":"ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"12587","article-title":"A two-stage framework in cross-spectrum domain for real-time speech enhancement","author":"Zhang","year":"2024"},{"key":"10.1016\/j.specom.2026.103395_b89","series-title":"ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"9281","article-title":"FRCRN: Boosting feature representation using frequency recurrence for monaural speech enhancement","author":"Zhao","year":"2022"},{"key":"10.1016\/j.specom.2026.103395_b90","article-title":"Sixty years of frequency-domain monaural speech enhancement: From traditional to deep learning methods","volume":"27","author":"Zheng","year":"2023","journal-title":"Trends Hear."}],"container-title":["Speech Communication"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167639326000439?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167639326000439?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T04:46:16Z","timestamp":1776919576000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0167639326000439"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5]]},"references-count":90,"alternative-id":["S0167639326000439"],"URL":"https:\/\/doi.org\/10.1016\/j.specom.2026.103395","relation":{},"ISSN":["0167-6393"],"issn-type":[{"value":"0167-6393","type":"print"}],"subject":[],"published":{"date-parts":[[2026,5]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"MaTSE: A hybrid Mamba-Transformer model for monaural Speech Enhancement","name":"articletitle","label":"Article Title"},{"value":"Speech Communication","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.specom.2026.103395","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"103395"}}