{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,24]],"date-time":"2025-06-24T04:04:30Z","timestamp":1750737870090,"version":"3.41.0"},"publisher-location":"Singapore","reference-count":34,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819665877","type":"print"},{"value":"9789819665884","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-96-6588-4_20","type":"book-chapter","created":{"date-parts":[[2025,6,23]],"date-time":"2025-06-23T14:40:48Z","timestamp":1750689648000},"page":"286-300","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Dual-Branch StarNet with\u00a0Mutual Attention and\u00a0U-Net Denoising for\u00a0Simultaneously Recognizing Keywords and\u00a0Speakers"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1018-1912","authenticated-orcid":false,"given":"Yuting","family":"He","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0559-7507","authenticated-orcid":false,"given":"Chengtai","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0305-2135","authenticated-orcid":false,"given":"Heng","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4619-6590","authenticated-orcid":false,"given":"Jianfeng","family":"Ren","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2855-9570","authenticated-orcid":false,"given":"Zheng","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6300-3503","authenticated-orcid":false,"given":"Heshan","family":"Du","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3831-3876","authenticated-orcid":false,"given":"Yinshui","family":"Xia","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,6,24]]},"reference":[{"key":"20_CR1","doi-asserted-by":"crossref","unstructured":"Bonet, D., et al.: Speech enhancement for wake-up-word detection in voice assistants. In: Proceedings of the IberSPEECH 2021, pp. 41\u201345 (2021)","DOI":"10.21437\/IberSPEECH.2021-9"},{"key":"20_CR2","doi-asserted-by":"crossref","unstructured":"Bovbjerg, H.S., Tan, Z.H.: Improving label-deficient keyword spotting through self-supervised pretraining. In: ICASSP, pp.\u00a01\u20135 (2023)","DOI":"10.1109\/ICASSPW59220.2023.10193371"},{"key":"20_CR3","doi-asserted-by":"crossref","unstructured":"Choi, S., et al.: Temporal convolution for real-time keyword spotting on mobile devices. In: Proceedings of the Interspeech 2019, pp. 3372\u20133376 (2019)","DOI":"10.21437\/Interspeech.2019-1363"},{"key":"20_CR4","doi-asserted-by":"crossref","unstructured":"Ding, K., Zong, M., Li, J., Li, B.: LETR: a lightweight and efficient transformer for keyword spotting. In: ICASSP, pp. 7987\u20137991 (2022)","DOI":"10.1109\/ICASSP43922.2022.9747295"},{"key":"20_CR5","doi-asserted-by":"publisher","first-page":"24013","DOI":"10.1007\/s11042-019-08293-7","volume":"79","author":"SA El-Moneim","year":"2020","unstructured":"El-Moneim, S.A., Nassar, M., Dessouky, M.I., Ismail, N.A., El-Fishawy, A.S., Abd El-Samie, F.E.: Text-independent speaker recognition using LSTM-RNN and speech enhancement. Multimed. Tools. Appl. 79, 24013\u201324028 (2020)","journal-title":"Multimed. Tools. Appl."},{"key":"20_CR6","doi-asserted-by":"publisher","first-page":"107882","DOI":"10.1016\/j.compeleceng.2022.107882","volume":"100","author":"S Farsiani","year":"2022","unstructured":"Farsiani, S., Izadkhah, H., Lotfi, S.: An optimum end-to-end text-independent speaker identification system using convolutional neural network. Comput. Electr. Eng. 100, 107882 (2022)","journal-title":"Comput. Electr. Eng."},{"key":"20_CR7","doi-asserted-by":"crossref","unstructured":"Hussain, S., Nguyen, V., Zhang, S., Visser, E.: Multi-task voice activated framework using self-supervised learning. In: ICASSP, pp. 6137\u20136141 (2022)","DOI":"10.1109\/ICASSP43922.2022.9746409"},{"key":"20_CR8","doi-asserted-by":"crossref","unstructured":"Jung, M., Jung, Y., Goo, J., Kim, H.: Multi-task network for noise-robust keyword spotting and speaker verification using CTC-based soft VAD and global query attention. In: Proceedings of the Interspeech 2020, pp. 931\u2013935 (2020)","DOI":"10.21437\/Interspeech.2020-1420"},{"key":"20_CR9","doi-asserted-by":"publisher","unstructured":"Kim, B., Chang, S., Lee, J., Sung, D.: Broadcasted residual learning for efficient keyword spotting. In: Proceedings of the Interspeech 2021, pp. 4538\u20134542 (2021). https:\/\/doi.org\/10.21437\/Interspeech.2021-383","DOI":"10.21437\/Interspeech.2021-383"},{"key":"20_CR10","unstructured":"Kingma, D.P., Welling, M.: Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114 (2013)"},{"key":"20_CR11","doi-asserted-by":"publisher","first-page":"2324","DOI":"10.1109\/TASLP.2024.3385277","volume":"32","author":"T Liu","year":"2024","unstructured":"Liu, T., Lee, K.A., Wang, Q., Li, H.: Golden Gemini is all you need: finding the sweet spots for speaker verification. IEEE\/ACM Trans. Audio Speech Lang. Process. 32, 2324\u20132337 (2024)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"20_CR12","doi-asserted-by":"crossref","unstructured":"Liu, X., Sahidullah, M., Kinnunen, T.: Learnable nonlinear compression for robust speaker verification. In: ICASSP, pp. 7962\u20137966 (2022)","DOI":"10.1109\/ICASSP43922.2022.9747185"},{"key":"20_CR13","doi-asserted-by":"crossref","unstructured":"L\u00f3pez-Espejo, I., Shekar, R.C., Tan, Z.H., Jensen, J., Hansen, J.H.: Filterbank learning for noise-robust small-footprint keyword spotting. In: ICASSP, pp.\u00a01\u20135 (2023)","DOI":"10.1109\/ICASSP49357.2023.10095436"},{"key":"20_CR14","doi-asserted-by":"publisher","first-page":"4169","DOI":"10.1109\/ACCESS.2021.3139508","volume":"10","author":"I L\u00f3pez-Espejo","year":"2021","unstructured":"L\u00f3pez-Espejo, I., Tan, Z.H., Hansen, J.H., Jensen, J.: Deep spoken keyword spotting: an overview. IEEE Access 10, 4169\u20134199 (2021)","journal-title":"IEEE Access"},{"key":"20_CR15","doi-asserted-by":"publisher","first-page":"2254","DOI":"10.1109\/TASLP.2021.3092567","volume":"29","author":"I L\u00f3pez-Espejo","year":"2021","unstructured":"L\u00f3pez-Espejo, I., Tan, Z.H., Jensen, J.: A novel loss function and training strategy for noise-robust keyword spotting. IEEE\/ACM Trans. Audio Speech Lang. Process 29, 2254\u20132266 (2021)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process"},{"key":"20_CR16","doi-asserted-by":"crossref","unstructured":"Ma, X., Dai, X., Bai, Y., Wang, Y., Fu, Y.: Rewrite the stars. arXiv preprint arXiv:2403.19967 (2024)","DOI":"10.1109\/CVPR52733.2024.00544"},{"key":"20_CR17","doi-asserted-by":"crossref","unstructured":"Mittermaier, S., K\u00fcrzinger, L., Waschneck, B., Rigoll, G.: Small-footprint keyword spotting on raw audio data with Sinc-convolutions. In: ICASSP, pp. 7454\u20137458 (2020)","DOI":"10.1109\/ICASSP40776.2020.9053395"},{"key":"20_CR18","doi-asserted-by":"crossref","unstructured":"M\u00f8rk, J., Bovbjerg, H.S., Kiss, G., Tan, Z.H.: Noise-robust keyword spotting through self-supervised pretraining. arXiv preprint arXiv:2403.18560 (2024)","DOI":"10.1109\/ICASSPW59220.2023.10193371"},{"key":"20_CR19","doi-asserted-by":"crossref","unstructured":"Ng, D., et al.: Small footprint multi-channel network for keyword spotting with centroid based awareness. In: Proceedings of the Interspeech, pp. 296\u2013300 (2023)","DOI":"10.21437\/Interspeech.2023-1210"},{"key":"20_CR20","doi-asserted-by":"publisher","first-page":"325","DOI":"10.1109\/TASLP.2023.3328283","volume":"32","author":"R Prabhavalkar","year":"2023","unstructured":"Prabhavalkar, R., Hori, T., Sainath, T.N., Schl\u00fcter, R., Watanabe, S.: End-to-end speech recognition: a survey. IEEE\/ACM Trans. Audio Speech Lang. Process. 32, 325\u2013351 (2023)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"20_CR21","unstructured":"Ruder, S.: An overview of multi-task learning in deep neural networks. arXiv preprint arXiv:1706.05098 (2017)"},{"key":"20_CR22","doi-asserted-by":"crossref","unstructured":"Sainath, T.N., Parada, C.: Convolutional neural networks for small-footprint keyword spotting. In: Interspeech, pp. 1478\u20131482 (2015)","DOI":"10.21437\/Interspeech.2015-352"},{"issue":"3","key":"20_CR23","doi-asserted-by":"publisher","first-page":"1839","DOI":"10.1007\/s00034-023-02542-9","volume":"43","author":"B Saritha","year":"2024","unstructured":"Saritha, B., Laskar, M.A., Kirupakaran, A.M., Laskar, R.H., Choudhury, M., Shome, N.: Deep learning-based end-to-end speaker identification using time-frequency representation of speech signal. Cir. Syst. Sig. Process. 43(3), 1839\u20131861 (2024)","journal-title":"Cir. Syst. Sig. Process."},{"issue":"26","key":"20_CR24","doi-asserted-by":"publisher","first-page":"18933","DOI":"10.1007\/s00521-023-08736-1","volume":"35","author":"N Shome","year":"2023","unstructured":"Shome, N., Saritha, B., Kashyap, R., Laskar, R.H.: A robust DNN model for text-independent speaker identification using non-speaker embeddings in diverse data conditions. Neural Comput. Appl. 35(26), 18933\u201318947 (2023)","journal-title":"Neural Comput. Appl."},{"key":"20_CR25","doi-asserted-by":"crossref","unstructured":"Tian, Y., Yao, H., Cai, M., Liu, Y., Ma, Z.: Improving RNN transducer modeling for small-footprint keyword spotting. In: ICASSP, pp. 5624\u20135628 (2021)","DOI":"10.1109\/ICASSP39728.2021.9414339"},{"key":"20_CR26","unstructured":"Upadhyay, R., Phlypo, R., Saini, R., Liwicki, M.: Sharing to learn and learning to share\u2013fitting together meta-learning, multi-task learning, and transfer learning: a meta review. arXiv preprint arXiv:2111.12146 (2021)"},{"issue":"3","key":"20_CR27","doi-asserted-by":"publisher","first-page":"247","DOI":"10.1016\/0167-6393(93)90095-3","volume":"12","author":"A Varga","year":"1993","unstructured":"Varga, A., Steeneken, H.: II. NOISEX-92: a database and an experiment to study the effect of additive noise on speech recognition systems. Speech Commun. 12(3), 247\u2013251 (1993)","journal-title":"Speech Commun."},{"key":"20_CR28","unstructured":"Vaswani, A., et al.: Attention is all you need. NeurIPS 30 (2017)"},{"key":"20_CR29","doi-asserted-by":"crossref","unstructured":"Wang, L., Gu, R., Zhuang, W., Gao, P., Wang, Y., Zou, Y.: Learning decoupling features through orthogonality regularization. In: ICASSP, pp. 7562\u20137566 (2022)","DOI":"10.1109\/ICASSP43922.2022.9747878"},{"key":"20_CR30","doi-asserted-by":"crossref","unstructured":"Wang, Z., Duan, S., Zeng, C., Yu, X., Yang, Y., Wu, H.: Robust speaker identification of IoT based on stacked sparse denoising auto-encoders. In: 2020 iThings and IEEE GreenCom CPSCom and IEEE SmartData and IEEE Cybermatics, pp. 252\u2013257 (2020)","DOI":"10.1109\/iThings-GreenCom-CPSCom-SmartData-Cybermatics50389.2020.00056"},{"key":"20_CR31","doi-asserted-by":"crossref","unstructured":"Yang, C., Saidutta, Y.M., Srinivasa, R.S., Lee, C.H., Shen, Y., Jin, H.: Robust keyword spotting for noisy environments by leveraging speech enhancement and speech presence probability. In: Proceedings of the Interspeech, pp. 1638\u20131642 (2023)","DOI":"10.21437\/Interspeech.2023-2222"},{"key":"20_CR32","doi-asserted-by":"crossref","unstructured":"Yang, S., Kim, B., Chung, I., Chang, S.: Personalized keyword spotting through multi-task learning. In: Proceedings of the Interspeech 2022, pp. 1881\u20131885 (2022)","DOI":"10.21437\/Interspeech.2022-947"},{"key":"20_CR33","doi-asserted-by":"crossref","unstructured":"Zhang, C., Yu, M., Weng, C., Yu, D.: Towards robust speaker verification with target speaker enhancement. In: ICASSP, pp. 6693\u20136697 (2021)","DOI":"10.1109\/ICASSP39728.2021.9414017"},{"key":"20_CR34","doi-asserted-by":"publisher","unstructured":"Zhang, Y., et al.: MFA-conformer: multi-scale feature aggregation conformer for automatic speaker verification. In: Proceedings of the Interspeech 2022, pp. 306\u2013310 (2022). https:\/\/doi.org\/10.21437\/Interspeech.2022-563","DOI":"10.21437\/Interspeech.2022-563"}],"container-title":["Lecture Notes in Computer Science","Neural Information Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-96-6588-4_20","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,23]],"date-time":"2025-06-23T14:40:54Z","timestamp":1750689654000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-96-6588-4_20"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9789819665877","9789819665884"],"references-count":34,"URL":"https:\/\/doi.org\/10.1007\/978-981-96-6588-4_20","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"24 June 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICONIP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Neural Information Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Auckland","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"New Zealand","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 December 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"6 December 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"31","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"iconip2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/iconip2024.org","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}