{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T01:16:30Z","timestamp":1783127790445,"version":"3.54.6"},"reference-count":52,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100002855","name":"Ministry of Science and Technology of the People&apos;s Republic of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100002855","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2023YFF1204102"],"award-info":[{"award-number":["2023YFF1204102"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Speech Communication"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1016\/j.specom.2026.103426","type":"journal-article","created":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T16:14:23Z","timestamp":1780676063000},"page":"103426","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["PRSE: A two-stage joint optimization approach for lightweight speech enhancement"],"prefix":"10.1016","volume":"182","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8880-449X","authenticated-orcid":false,"given":"Haixin","family":"Guan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guangyong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yanhua","family":"Long","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiaen","family":"Liang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaobin","family":"Tan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"2","key":"10.1016\/j.specom.2026.103426_b1","doi-asserted-by":"crossref","first-page":"113","DOI":"10.1109\/TASSP.1979.1163209","article-title":"Suppression of acoustic noise in speech using spectral subtraction","volume":"27","author":"Boll","year":"1979","journal-title":"IEEE Trans. Acoust. Speech Signal Process."},{"issue":"2","key":"10.1016\/j.specom.2026.103426_b2","first-page":"345","article-title":"Elimination of the musical noise phenomenon with the Ephraim and Malah noise suppressor","volume":"2","author":"Capp\u00e9","year":"1994","journal-title":"IEEE\/ACM Trans. Acoust. Speech Signal Process."},{"key":"10.1016\/j.specom.2026.103426_b3","doi-asserted-by":"crossref","unstructured":"Chen, J., Shi, Y., Liu, W., Rao, W., He, S., Li, A., Wang, Y., Wu, Z., Shang, S., Zheng, C., 2023. Gesper: A Unified Framework for General Speech Restoration. In: Proc. ICASSP. pp. 1\u20132.","DOI":"10.1109\/ICASSP49357.2023.10095557"},{"key":"10.1016\/j.specom.2026.103426_b4","series-title":"Proc. ICASSP","first-page":"7857","article-title":"Fullsubnet+: Channel attention fullsubnet with complex spectrograms for speech enhancement","author":"Chen","year":"2022"},{"key":"10.1016\/j.specom.2026.103426_b5","doi-asserted-by":"crossref","first-page":"662","DOI":"10.1109\/OJSP.2024.3376293","article-title":"ICASSP 2023 speech signal improvement challenge","volume":"5","author":"Cutler","year":"2024","journal-title":"IEEE Open J. Signal Process."},{"issue":"6","key":"10.1016\/j.specom.2026.103426_b6","doi-asserted-by":"crossref","first-page":"1109","DOI":"10.1109\/TASSP.1984.1164453","article-title":"Speech enhancement using a minimum mean-square error log-spectral amplitude estimator","volume":"32","author":"Ephraim","year":"1984","journal-title":"IEEE\/ACM Trans. Acoust. Speech Signal Process."},{"issue":"2","key":"10.1016\/j.specom.2026.103426_b7","doi-asserted-by":"crossref","first-page":"443","DOI":"10.1109\/TASSP.1985.1164550","article-title":"Speech enhancement using a minimummean square error short-time spectral amplitude estimator","volume":"33","author":"Ephraim","year":"1985","journal-title":"IEEE\/ACM Trans. Acoust. Speech Signal Process."},{"key":"10.1016\/j.specom.2026.103426_b8","series-title":"Speech Processing, Transmission and Quality Aspects (STQ); Distributed Speech Recognition; Front-End Feature Extraction Algorithm; Compression Algorithms","author":"ETSI","year":"2003"},{"key":"10.1016\/j.specom.2026.103426_b9","series-title":"Proc. APSIPA","first-page":"559","article-title":"Low-Power convolutional recurrent neural network for monaural speech enhancement","author":"Gao","year":"2021"},{"issue":"4","key":"10.1016\/j.specom.2026.103426_b10","first-page":"1383","article-title":"Unbiased MMSE-based noise power estimation with low complexity and low tracking delay","volume":"20","author":"Gerkmann","year":"2012","journal-title":"IEEE\/ACM Trans. Acoust. Speech Signal Process."},{"key":"10.1016\/j.specom.2026.103426_b11","doi-asserted-by":"crossref","unstructured":"Gholami, B., El-Khamy, M., Song, K.-B., 2024. Knowledge Distillation for Tiny Speech Enhancement with Latent Feature Augmentation. In: Proc. INTERSPEECH. pp. 652\u2013656.","DOI":"10.21437\/Interspeech.2024-1383"},{"key":"10.1016\/j.specom.2026.103426_b12","doi-asserted-by":"crossref","unstructured":"Guan, H., Dai, W., Wang, G., Tan, X., Li, P., Liang, J., 2024. Reducing Speech Distortion and Artifacts for Speech Enhancement by Loss Function. In: Proc. INTERSPEECH. pp. 1730\u20131734.","DOI":"10.21437\/Interspeech.2024-620"},{"key":"10.1016\/j.specom.2026.103426_b13","series-title":"Proc. ICASSP","first-page":"6633","article-title":"Fullsubnet: A full-band and sub-band fusion model for real-time single-channel speech enhancement","author":"Hao","year":"2021"},{"key":"10.1016\/j.specom.2026.103426_b14","doi-asserted-by":"crossref","unstructured":"Hu, Y., Liu, Y., et al., 2020. DCCRN: Deep complex convolution recurrent network for phase-aware speech enhancement. In: Proc. INTERSPEECH. pp. 2472\u20132476.","DOI":"10.21437\/Interspeech.2020-2537"},{"key":"10.1016\/j.specom.2026.103426_b15","series-title":"ITU-T Recommendation P.56: Objective Measurement of Active Speech Level","author":"ITU-T","year":"1994"},{"key":"10.1016\/j.specom.2026.103426_b16","series-title":"ITU-T Recommendation P.808: Subjective Evaluation of Speech Quality with a Crowdsourcing Approach","author":"ITU-T","year":"2018"},{"key":"10.1016\/j.specom.2026.103426_b17","series-title":"Alternating approach-putt models for multi-stage speech enhancement","author":"Jeong","year":"2025"},{"key":"10.1016\/j.specom.2026.103426_b18","first-page":"825","article-title":"On loss functions for supervised monaural time-domain speech enhancement","volume":"28","author":"Kolb\u00e6k","year":"2020","journal-title":"IEEE\/ACM Trans. Acoust. Speech Signal Process."},{"key":"10.1016\/j.specom.2026.103426_b19","series-title":"Proc. ICASSP","first-page":"626","article-title":"SDR\u2013half-baked or well done?","author":"Le Roux","year":"2019"},{"key":"10.1016\/j.specom.2026.103426_b20","series-title":"Real-time monaural speech enhancement with short-time discrete cosine transform","author":"Li","year":"2021"},{"key":"10.1016\/j.specom.2026.103426_b21","doi-asserted-by":"crossref","unstructured":"Li, A., Liu, W., Luo, X., Zheng, C., Li, X., 2021. ICASSP 2021 Deep Noise Suppression Challenge: Decoupling Magnitude and Phase Optimization with a Two-Stage Deep Network. In: Proc. ICASSP. pp. 6628\u20136632.","DOI":"10.1109\/ICASSP39728.2021.9414062"},{"key":"10.1016\/j.specom.2026.103426_b22","unstructured":"Li, A., Yin, Z., Wang, T., Fang, Q., Hu, F., Qian, Y., 2004. RASC863 - A Chinese Speech Corpus with Four Regional Accents. In: Proceedings of ICSLT-O-COCOSDA. Vol. 2, pp. 15\u201319."},{"key":"10.1016\/j.specom.2026.103426_b23","doi-asserted-by":"crossref","unstructured":"Liu, L., Guan, H., Ma, J., Dai, W., Wang, G., Ding, S., 2023. A Mask Free Neural Network for Monaural Speech Enhancement. In: Proc. INTERSPEECH. pp. 2468\u20132472.","DOI":"10.21437\/Interspeech.2023-339"},{"key":"10.1016\/j.specom.2026.103426_b24","doi-asserted-by":"crossref","DOI":"10.1016\/j.neunet.2025.107562","article-title":"Explicit estimation of magnitude and phase spectra in parallel for high-quality speech enhancement","volume":"189","author":"Lu","year":"2025","journal-title":"Neural Netw."},{"issue":"8","key":"10.1016\/j.specom.2026.103426_b25","first-page":"1256","article-title":"Conv-tasnet: Surpassing ideal timefrequency magnitude masking for speech separation","volume":"27","author":"Luo","year":"2019","journal-title":"IEEE\/ACM Trans. Acoust. Speech Signal Process."},{"issue":"11","key":"10.1016\/j.specom.2026.103426_b26","doi-asserted-by":"crossref","first-page":"1680","DOI":"10.1109\/LSP.2018.2871419","article-title":"A deep learning loss function based on the perceptual evaluation of the speech quality","volume":"25","author":"Martin-Donas","year":"2018","journal-title":"Signal Process. Lett."},{"key":"10.1016\/j.specom.2026.103426_b27","series-title":"Proc. ICASSP","first-page":"7092","article-title":"Ideal ratio mask estimation using deep neural networks for robust speech recognition","author":"Narayanan","year":"2013"},{"key":"10.1016\/j.specom.2026.103426_b28","doi-asserted-by":"crossref","first-page":"1485","DOI":"10.1109\/LSP.2020.3016837","article-title":"On mean absolute error for deep neural network based vector-to-vector regression","volume":"27","author":"Qi","year":"2020","journal-title":"Signal Process. Lett."},{"key":"10.1016\/j.specom.2026.103426_b29","doi-asserted-by":"crossref","unstructured":"Reddy, C.K.A., Dubey, H., Gopal, V., Cutler, R., Braun, S., Gamper, H., Aichner, R., Srinivasan, S., 2021a. ICASSP 2021 Deep Noise Suppression Challenge. In: Proc. ICASSP. pp. 6623\u20136627.","DOI":"10.1109\/ICASSP39728.2021.9415105"},{"key":"10.1016\/j.specom.2026.103426_b30","series-title":"ICASSP","first-page":"6493","article-title":"DNSMOS: A non-intrusive perceptual objective speech quality metric to evaluate noise suppressors","author":"Reddy","year":"2021"},{"key":"10.1016\/j.specom.2026.103426_b31","doi-asserted-by":"crossref","unstructured":"Reddy, C.K., Gopal, V., Cutler, R., Beyrami, E., Cheng, R., Dubey, H., Matusevych, S., Aichner, R., Aazami, A., Braun, S., Rana, P., Srinivasan, S., Gehrke, J., 2020. The INTERSPEECH 2020 Deep Noise Suppression Challenge: Datasets, Subjective Testing Framework, and Challenge Results. In: Proc. INTERSPEECH. pp. 2492\u20132496.","DOI":"10.21437\/Interspeech.2020-3038"},{"key":"10.1016\/j.specom.2026.103426_b32","series-title":"Proc. ICASSP","first-page":"749","article-title":"Perceptual evaluation of speech quality (PESQ)-a new method for speech quality assessment of telephone networks and codecs","volume":"Vol. 2","author":"Rix","year":"2001"},{"key":"10.1016\/j.specom.2026.103426_b33","series-title":"Proc. ICASSP","first-page":"971","article-title":"GTCRN: A speech enhancement model requiring ultralow computational resources","author":"Rong","year":"2024"},{"key":"10.1016\/j.specom.2026.103426_b34","series-title":"P.808 Multilingual Speech Enhancement Testing: approach and results of URGENT 2025 challenge","author":"Sach","year":"2025"},{"key":"10.1016\/j.specom.2026.103426_b35","series-title":"Weight, block or unit? Exploring sparsity tradeoffs for speech enhancement on tiny neural accelerators","author":"Stamenovic","year":"2021"},{"issue":"3","key":"10.1016\/j.specom.2026.103426_b36","doi-asserted-by":"crossref","first-page":"185","DOI":"10.1121\/1.1915893","article-title":"A scale for the measurement of the psychological magnitude pitch","volume":"8","author":"Stevens","year":"1937","journal-title":"J. Acoust. Soc. Am."},{"issue":"7","key":"10.1016\/j.specom.2026.103426_b37","doi-asserted-by":"crossref","first-page":"2125","DOI":"10.1109\/TASL.2011.2114881","article-title":"An algorithm for intelligibility prediction of time\u2013frequency weighted noisy speech","volume":"9","author":"Taal","year":"2011","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.specom.2026.103426_b38","doi-asserted-by":"crossref","unstructured":"Tan, K., Wang, D.L., 2018. A convolutional recurrent neural network for real-time speech enhancement. In: Proc. INTERSPEECH. pp. 3229\u20133233.","DOI":"10.21437\/Interspeech.2018-1405"},{"key":"10.1016\/j.specom.2026.103426_b39","doi-asserted-by":"crossref","first-page":"1785","DOI":"10.1109\/TASLP.2021.3082282","article-title":"Towards model compression for deep learning based speech enhancement","volume":"29","author":"Tan","year":"2021","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.specom.2026.103426_b40","doi-asserted-by":"crossref","unstructured":"Tashev, I., Lovitt, A., Acero, A., 2009. Unified framework for single channel speech enhancement. In: 2009 IEEE Pacific Rim Conference on Communications, Computers and Signal Processing. pp. 883\u2013888.","DOI":"10.1109\/PACRIM.2009.5291253"},{"key":"10.1016\/j.specom.2026.103426_b41","series-title":"Proc. APSIPA","first-page":"131","article-title":"Investigating low-distortion speech enhancement with discrete cosine transform features for robust speech recognition","author":"Tsao","year":"2022"},{"key":"10.1016\/j.specom.2026.103426_b42","series-title":"International Workshop on Multimedia Signal Processing","first-page":"1","article-title":"A hybrid DSP\/Deep learning approach to real-time full-band speech enhancement","author":"Valin","year":"2018"},{"issue":"3","key":"10.1016\/j.specom.2026.103426_b43","doi-asserted-by":"crossref","first-page":"247","DOI":"10.1016\/0167-6393(93)90095-3","article-title":"Assessment for automatic speech recognition: Ii. NOISEX-92: A database and an experiment to study the effect of additive noise on speech recognition systems","volume":"12","author":"Varga","year":"1993","journal-title":"Speech Commun."},{"key":"10.1016\/j.specom.2026.103426_b44","doi-asserted-by":"crossref","first-page":"3221","DOI":"10.1109\/TASLP.2023.3304482","article-title":"TFGridNet: Integrating full-and sub-band modeling for speech separation","volume":"31","author":"Wang","year":"2023","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.specom.2026.103426_b45","series-title":"Proc. ICASSP","first-page":"1","article-title":"ZipEnhancer: Dual-path down-up sampling-based zipformer for monaural speech enhancement","author":"Wang","year":"2025"},{"key":"10.1016\/j.specom.2026.103426_b46","series-title":"Proc. ICASSP","first-page":"486","article-title":"Multi-microphone complex spectral mapping for speech dereverberation","author":"Wang","year":"2020"},{"key":"10.1016\/j.specom.2026.103426_b47","series-title":"Proc. of 11th Workshop on Statistical Signal Processing","first-page":"496","article-title":"Simple alternatives to the Ephraim and Malah suppression rule for speech enhancement","author":"Wolfe","year":"2001"},{"key":"10.1016\/j.specom.2026.103426_b48","series-title":"Proc. ICASSP","first-page":"1","article-title":"LiSenNet: lightweight sub-band and dual-path modeling for real-time speech enhancement","author":"Yan","year":"2025"},{"key":"10.1016\/j.specom.2026.103426_b49","series-title":"Proc. ICASSP","first-page":"10671","article-title":"FSPEN: an ultra-lightweight network for real time speech enahncment","author":"Yang","year":"2024"},{"issue":"2","key":"10.1016\/j.specom.2026.103426_b50","doi-asserted-by":"crossref","first-page":"358","DOI":"10.1016\/j.specom.2012.09.004","article-title":"Optimization and evaluation of sigmoid function with a priori SNR estimate for real-time speech enhancement","volume":"55","author":"Yong","year":"2013","journal-title":"Speech Commun."},{"key":"10.1016\/j.specom.2026.103426_b51","doi-asserted-by":"crossref","unstructured":"Yu, J., Chen, H., Luo, Y., Gu, R., Weng, C., 2023. High fidelity speech enhancement with band-split RNN. In: Proc. INTERSPEECH. pp. 2483\u20132487.","DOI":"10.21437\/Interspeech.2023-1433"},{"key":"10.1016\/j.specom.2026.103426_b52","doi-asserted-by":"crossref","unstructured":"Zhang, W., Saijo, K., Jung, J.w., Li, C., Watanabe, S., Qian, Y., 2024. Beyond Performance Plateaus: A Comprehensive Study on Scalability in Speech Enhancement. In: Proc. INTERSPEECH. pp. 1740\u20131744.","DOI":"10.21437\/Interspeech.2024-1266"}],"container-title":["Speech Communication"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167639326000749?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167639326000749?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T00:23:07Z","timestamp":1783124587000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0167639326000749"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":52,"alternative-id":["S0167639326000749"],"URL":"https:\/\/doi.org\/10.1016\/j.specom.2026.103426","relation":{},"ISSN":["0167-6393"],"issn-type":[{"value":"0167-6393","type":"print"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"PRSE: A two-stage joint optimization approach for lightweight speech enhancement","name":"articletitle","label":"Article Title"},{"value":"Speech Communication","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.specom.2026.103426","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"103426"}}