{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,28]],"date-time":"2025-03-28T01:53:30Z","timestamp":1743126810754,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":25,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789819787487"},{"type":"electronic","value":"9789819787494"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-981-97-8749-4_13","type":"book-chapter","created":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T15:57:22Z","timestamp":1730303842000},"page":"175-186","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Speech Emotion Recognition Using U-Net"],"prefix":"10.1007","author":[{"given":"Yongzhen","family":"Yu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Daming","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,10,31]]},"reference":[{"issue":"4","key":"13_CR1","doi-asserted-by":"publisher","first-page":"2525","DOI":"10.1007\/s11277-023-10244-3","volume":"129","author":"MJ Al-Dujaili","year":"2023","unstructured":"Al-Dujaili, M.J., Ebrahimi-Moghadam, A.: Speech emotion recognition: a comprehensive survey. Wireless Pers. Commun. 129(4), 2525\u20132561 (2023)","journal-title":"Wireless Pers. Commun."},{"key":"13_CR2","doi-asserted-by":"crossref","unstructured":"Schuller, B., Rigoll, G., Lang, M.: Hidden Markov model-based speech emotion recognition. In: 2003 IEEE International Conference on Acoustics, Speech, and Signal Processing (2003)","DOI":"10.1109\/ICME.2003.1220939"},{"key":"13_CR3","doi-asserted-by":"crossref","unstructured":"Hu, H., Xu, M.X., Wu, W.: GMM supervector based SVM with spectral features for speech emotion recognition. In: 2007 IEEE International Conference on Acoustics, Speech and Signal Processing-ICASSP'07, vol. 4, pp. IV-413\u2013IV-416. IEEE (2007)","DOI":"10.1109\/ICASSP.2007.366937"},{"key":"13_CR4","doi-asserted-by":"crossref","unstructured":"Anvarjon, T., Mustaqeem, K., Won, S.: Deep-net: a lightweight CNN-based speech emotion recognition system using deep frequency features. Sensors 20(18), 5212 (2020)","DOI":"10.3390\/s20185212"},{"key":"13_CR5","doi-asserted-by":"publisher","DOI":"10.1016\/j.asoc.2021.107101","volume":"102","author":"S Kwon","year":"2021","unstructured":"Kwon, S.: Att-Net: enhanced emotion recognition system using lightweight self-attention module. Appl. Soft Comput. 102, 107101 (2021)","journal-title":"Appl. Soft Comput."},{"key":"13_CR6","doi-asserted-by":"crossref","unstructured":"Meyer, P., Xu, Z., Fingscheidt, T.: Improving convolutional recurrent neural networks for speech emotion recognition. In: 2021 IEEE Spoken Language Technology Workshop (SLT), pp. 365\u2013372. IEEE (2021)","DOI":"10.1109\/SLT48900.2021.9383513"},{"key":"13_CR7","doi-asserted-by":"crossref","unstructured":"Liu, J., Liu, Z., Wang, L., et al.: Temporal attention convolutional network for speech emotion recognition with latent representation. In: INTERSPEECH, pp. 2337\u20132341 (2020)","DOI":"10.21437\/Interspeech.2020-1520"},{"key":"13_CR8","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2023.110525","volume":"270","author":"K Mustaqeem","year":"2023","unstructured":"Mustaqeem, K., El Saddik, A., Alotaibi, F.S., et al.: AAD-Net: advanced end-to-end signal processing system for human emotion detection & recognition using attention-based deep echo state network. Knowl.-Based Syst. 270, 110525 (2023)","journal-title":"Knowl.-Based Syst."},{"key":"13_CR9","unstructured":"Stoller, D., Ewert, S., Dixon, S.: Wave-u-net: A multi-scale neural network for end-to-end audio source separation. arXiv preprint arXiv:1806.03185 (2018)"},{"key":"13_CR10","doi-asserted-by":"crossref","unstructured":"Pandey, A., Wang, D.L.: TCNN: temporal convolutional neural network for real-time speech enhancement in the time domain. In: ICASSP 2019\u20132019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6875\u20136879. IEEE (2019)","DOI":"10.1109\/ICASSP.2019.8683634"},{"key":"13_CR11","doi-asserted-by":"crossref","unstructured":"Giri, R., Isik, U., Krishnaswamy, A.: Attention wave-u-net for speech enhancement. In: 2019 IEEE Workshop on Applications of Signal Processing to Audio and Acoustics (WASPAA), pp. 249\u2013253. IEEE (2019)","DOI":"10.1109\/WASPAA.2019.8937186"},{"key":"13_CR12","unstructured":"Oord, A., Dieleman, S., Zen, H., et al.: Wavenet: a generative model for raw audio. arXiv preprint arXiv:1609.03499 (2016)"},{"key":"13_CR13","unstructured":"Glorot, X., Bordes, A., Bengio, Y.: Deep sparse rectifier neural networks. In: Proceedings of the Fourteenth International Conference on Artificial Intelligence and Statistics. JMLR Workshop and Conference Proceedings, pp. 315\u2013323 (2011)"},{"key":"13_CR14","doi-asserted-by":"crossref","unstructured":"Lu, Z., Li, J., Liu, H., et al.: Transformer for single image super-resolution. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 457\u2013466 (2022)","DOI":"10.1109\/CVPRW56347.2022.00061"},{"key":"13_CR15","unstructured":"Dumoulin, V., Visin, F.: A guide to convolution arithmetic for deep learning. arXiv preprint arXiv:1603.07285 (2016)"},{"issue":"5","key":"13_CR16","doi-asserted-by":"publisher","first-page":"710","DOI":"10.1109\/TIP.2004.826093","volume":"13","author":"T Blu","year":"2004","unstructured":"Blu, T., Th\u00e9venaz, P., Unser, M.: Linear interpolation revitalized. IEEE Trans. Image Process. 13(5), 710\u2013719 (2004)","journal-title":"IEEE Trans. Image Process."},{"key":"13_CR17","doi-asserted-by":"publisher","unstructured":"Guan, H., et al.: Road marking extraction in UAV imagery using attentive capsule feature pyramid network. Int. J. Appl. Earth Observ. Geoinform. 107, 102677 (2022). ISSN 1569-8432. https:\/\/doi.org\/10.1016\/j.jag.2022.102677","DOI":"10.1016\/j.jag.2022.102677"},{"key":"13_CR18","doi-asserted-by":"crossref","unstructured":"Huang, S., Lu, Z., Cheng, R., He, C.: FaPN: feature-aligned pyramid network for dense image prediction.\u00a0In: 2021 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 844\u2013853 (2021)","DOI":"10.1109\/ICCV48922.2021.00090"},{"key":"13_CR19","doi-asserted-by":"publisher","unstructured":"Wen, Y., Zhang, K., Li, Z., et al.: A discriminative feature learning approach for deep face recognition. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) Computer Vision \u2013 ECCV 2016. ECCV 2016. LNCS, vol. 9911. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46478-7_31","DOI":"10.1007\/978-3-319-46478-7_31"},{"key":"13_CR20","unstructured":"Loshchilov I, Hutter F. Fixing weight decay regularization in adam[J]. 2018"},{"key":"13_CR21","doi-asserted-by":"publisher","first-page":"913","DOI":"10.1007\/s12652-016-0406-z","volume":"8","author":"Y Li","year":"2017","unstructured":"Li, Y., Tao, J., Chao, L., et al.: CHEAVD: a Chinese natural emotional audio\u2013visual database. J. Ambient. Intell. Humaniz. Comput. 8, 913\u2013924 (2017)","journal-title":"J. Ambient. Intell. Humaniz. Comput."},{"key":"13_CR22","doi-asserted-by":"crossref","unstructured":"Burkhardt, F., Paeschke, A., Rolfes, M., et al.: A database of German emotional speech. In: Interspeech, vol. 5, pp. 1517\u20131520 (2005)","DOI":"10.21437\/Interspeech.2005-446"},{"issue":"5","key":"13_CR23","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0196391","volume":"13","author":"SR Livingstone","year":"2018","unstructured":"Livingstone, S.R., Russo, F.A.: The ryerson audio-visual database of emotional speech and song (RAVDESS): a dynamic, multimodal set of facial and vocal expressions in North American English. PLoS ONE 13(5), e0196391 (2018)","journal-title":"PLoS ONE"},{"key":"13_CR24","doi-asserted-by":"crossref","unstructured":"Liu, Z., Kang, X., Ren, F.: Dual-TBNet: improving the robustness of speech features via dual-Transformer-BiLSTM for speech emotion recognition. IEEE\/ACM Trans. Audio, Speech Lang. Process. (2023)","DOI":"10.1109\/TASLP.2023.3282092"},{"issue":"21","key":"13_CR25","doi-asserted-by":"publisher","first-page":"9897","DOI":"10.3390\/app11219897","volume":"11","author":"H Zhang","year":"2021","unstructured":"Zhang, H., Huang, H., Han, H.: A novel heterogeneous parallel convolution BI-LSTM for speech emotion recognition. Appl. Sci. 11(21), 9897 (2021)","journal-title":"Appl. Sci."}],"container-title":["Communications in Computer and Information Science","Data Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-97-8749-4_13","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T16:13:23Z","timestamp":1730304803000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-97-8749-4_13"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9789819787487","9789819787494"],"references-count":25,"URL":"https:\/\/doi.org\/10.1007\/978-981-97-8749-4_13","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"type":"print","value":"1865-0929"},{"type":"electronic","value":"1865-0937"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"31 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICPCSEE","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference of Pioneering Computer Scientists, Engineers and Educators","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Macao","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30 September 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"10","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icpcsee2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/2024.icpcsee.org","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}