{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T22:15:47Z","timestamp":1782512147556,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":34,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,30]]},"DOI":"10.1145\/3731715.3733391","type":"proceedings-article","created":{"date-parts":[[2025,6,25]],"date-time":"2025-06-25T18:31:04Z","timestamp":1750876264000},"page":"625-633","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["MixSENet: A Lightweight Model for Speech Enhancement with Multi-Scale Features and Contextual Modeling"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-1926-5116","authenticated-orcid":false,"given":"ChuiKe","family":"Kong","sequence":"first","affiliation":[{"name":"College of Intelligent Equipment, Shandong University of Science and Technology, Taian, Shandong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5461-8901","authenticated-orcid":false,"given":"Guangcun","family":"Wei","sequence":"additional","affiliation":[{"name":"College of Intelligent Equipment, Shandong University of Science and Technology, Taian, Shandong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-8684-2450","authenticated-orcid":false,"given":"Shuo","family":"Li","sequence":"additional","affiliation":[{"name":"College of Intelligent Equipment, Shandong University of Science and Technology, Taian, Shandong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-5267-5111","authenticated-orcid":false,"given":"Penghao","family":"Ma","sequence":"additional","affiliation":[{"name":"College of Intelligent Equipment, Shandong University of Science and Technology, Taian, Shandong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3774-2262","authenticated-orcid":false,"given":"Changhao","family":"Li","sequence":"additional","affiliation":[{"name":"College of Intelligent Equipment, Shandong University of Science and Technology, Taian, Shandong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,6,30]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Conv-tasnet: Surpassing ideal time--frequency magnitude masking for speech separation","author":"Luo Yi","year":"2019","unstructured":"Yi Luo and Nima Mesgarani. Conv-tasnet: Surpassing ideal time--frequency magnitude masking for speech separation. IEEE\/ACM transactions on audio, speech, and language processing, 27(8):1256--1266, 2019."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054536"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Neil Shah Hemant A Patil and Meet H Soni. Time-frequency mask-based speech enhancement using convolutional generative adversarial network. In 2018 Asia-Pacific Signal and Information Processing Association Annual Summit and Conference (APSIPA ASC) pages 1246--1251. IEEE 2018.","DOI":"10.23919\/APSIPA.2018.8659692"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746171"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746273"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-815"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-1084"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747120"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-1017"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2409"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747521"},{"key":"e_1_3_2_1_12_1","first-page":"361","volume-title":"INTERSPEECH","author":"Li Nan","year":"2022","unstructured":"Nan Li, Meng Ge, Longbiao Wang, Masashi Unoki, Sheng Li, and Jianwu Dang. Global signal-to-noise ratio estimation based on multi-subband processing using convolutional neural network. In INTERSPEECH, pages 361--365, 2022."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2021.3128374"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2021.3093859"},{"issue":"11","key":"e_1_3_2_1_15_1","first-page":"3607","article-title":"Speech enhancement algorithm based on multi-scale ladder-type time-frequency conformer gan","volume":"43","author":"Yutang JIN, Yisong WANG, Lihui","year":"2023","unstructured":"Yutang JIN, Yisong WANG, Lihui WANG, and Pengli ZHAO. Speech enhancement algorithm based on multi-scale ladder-type time-frequency conformer gan. Journal of Computer Applications, 43(11):3607, 2023.","journal-title":"Journal of Computer Applications"},{"key":"e_1_3_2_1_16_1","volume-title":"Dbt-net: Dual-branch federative magnitude and phase estimation with attention-in-attention transformer for monaural speech enhancement","author":"Yu Guochen","year":"2022","unstructured":"Guochen Yu, Andong Li, Hui Wang, Yutian Wang, Yuxuan Ke, and Chengshi Zheng. Dbt-net: Dual-branch federative magnitude and phase estimation with attention-in-attention transformer for monaural speech enhancement. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 30:2629--2644, 2022."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3064421"},{"key":"e_1_3_2_1_18_1","volume-title":"A time-frequency attention module for neural speech enhancement","author":"Zhang Qiquan","year":"2022","unstructured":"Qiquan Zhang, Xinyuan Qian, Zhaoheng Ni, Aaron Nicolson, Eliathamby Ambikairajah, and Haizhou Li. A time-frequency attention module for neural speech enhancement. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 31:462--475, 2022."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2913512"},{"key":"e_1_3_2_1_20_1","volume-title":"Muse: Flexible voiceprint receptive fields and multi-path fusion enhanced taylor transformer for u-net-based speech enhancement. arXiv preprint arXiv:2406.04589","author":"Lin Zizhen","year":"2024","unstructured":"Zizhen Lin, Xiaoting Chen, and Junyu Wang. Muse: Flexible voiceprint receptive fields and multi-path fusion enhanced taylor transformer for u-net-based speech enhancement. arXiv preprint arXiv:2406.04589, 2024."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01176"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN60899.2024.10651326"},{"key":"e_1_3_2_1_24_1","unstructured":"Alexandre D\u00e9fossez Nicolas Usunier L\u00e9on Bottou and Francis Bach. Demucs: Deep extractor for music sources with extra unlabeled data remixed."},{"key":"e_1_3_2_1_25_1","first-page":"2736","volume-title":"Interspeech","author":"Kim Eesung","year":"2021","unstructured":"Eesung Kim and Hyeji Seo. Se-conformer: Time-domain speech enhancement using conformer. In Interspeech, pages 2736--2740, 2021."},{"key":"e_1_3_2_1_26_1","volume-title":"Metricgan: An improved version of metricgan for speech enhancement","author":"Fu Szu-Wei","year":"2021","unstructured":"Szu-Wei Fu, Cheng Yu, Tsun-An Hsieh, Peter Plantinga, Mirco Ravanelli, Xugang Lu, and Yu Tsao. Metricgan: An improved version of metricgan for speech enhancement. 2021."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413740"},{"key":"e_1_3_2_1_28_1","first-page":"31521","volume-title":"International Conference on Machine Learning","author":"Shin Wooseok","year":"2023","unstructured":"Wooseok Shin, Byung Hoon Lee, Jin Sob Kim, Hyun Joon Park, and Sung Won Han. Metricgan-okd: multi-metric optimization of metricgan via online knowledge distillation for speech enhancement. In International Conference on Machine Learning, pages 31521--31538. PMLR, 2023."},{"key":"e_1_3_2_1_29_1","volume-title":"Jin Sob Kim, Byung Hoon Lee, and Sung Won Han. Multi-view attention transfer for efficient speech enhancement.","author":"Shin Wooseok","year":"2022","unstructured":"Wooseok Shin, Hyun Joon Park, Jin Sob Kim, Byung Hoon Lee, and Sung Won Han. Multi-view attention transfer for efficient speech enhancement. 2022."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2016-24"},{"key":"e_1_3_2_1_31_1","volume-title":"Adam: A method for stochastic optimization. (No Title)","author":"Diederik P Kingma","year":"2014","unstructured":"P Kingma Diederik. Adam: A method for stochastic optimization. (No Title), 2014."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2001.941023"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2007.911054"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2011.2114881"}],"event":{"name":"ICMR '25: International Conference on Multimedia Retrieval","location":"Chicago IL USA","acronym":"ICMR '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 2025 International Conference on Multimedia Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3731715.3733391","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T04:06:18Z","timestamp":1755749178000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3731715.3733391"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,30]]},"references-count":34,"alternative-id":["10.1145\/3731715.3733391","10.1145\/3731715"],"URL":"https:\/\/doi.org\/10.1145\/3731715.3733391","relation":{},"subject":[],"published":{"date-parts":[[2025,6,30]]},"assertion":[{"value":"2025-06-30","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}