{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T17:46:32Z","timestamp":1776879992612,"version":"3.51.2"},"publisher-location":"Singapore","reference-count":27,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819620531","type":"print"},{"value":"9789819620548","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-96-2054-8_2","type":"book-chapter","created":{"date-parts":[[2025,1,2]],"date-time":"2025-01-02T15:45:26Z","timestamp":1735832726000},"page":"16-29","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["A Multi-aspect Multi-granularity Pronunciation Assessment Method Based on\u00a0Branchformer Encoder and\u00a0Hierarchical Aggregation"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-3241-365X","authenticated-orcid":false,"given":"Wenxu","family":"Du","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1681-1089","authenticated-orcid":false,"given":"Aishan","family":"Wumaier","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yahui","family":"Shi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nian","family":"Yi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dehua","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,1,3]]},"reference":[{"issue":"2","key":"2_CR1","doi-asserted-by":"publisher","first-page":"269","DOI":"10.1109\/TLT.2020.2980261","volume":"13","author":"C Tejedor-Garc\u00eda","year":"2020","unstructured":"Tejedor-Garc\u00eda, C., Escudero-Mancebo, D., C\u00e1mara-Arenas, E., Gonz\u00e1lez-Ferreras, C., Carde\u00f1oso-Payo, V.: Assessing pronunciation improvement in students of English using a controlled computer-assisted pronunciation tool. IEEE Trans. Learn. Technol. 13(2), 269\u2013282 (2020)","journal-title":"IEEE Trans. Learn. Technol."},{"key":"2_CR2","doi-asserted-by":"crossref","unstructured":"Wang, Y.B., Lee, L.S.: Improved approaches of modeling and detecting error patterns with empirical analysis for computer-aided pronunciation training. In: 2012 IEEE International Conference on acoustics, speech and signal processing (ICASSP), pp. 5049\u20135052. IEEE (2012)","DOI":"10.1109\/ICASSP.2012.6289055"},{"key":"2_CR3","doi-asserted-by":"crossref","unstructured":"Shi, J., Huo, N., Jin, Q.: Context-aware goodness of pronunciation for computer-assisted pronunciation training. arXiv preprint arXiv:2008.08647 (2020)","DOI":"10.21437\/Interspeech.2020-2953"},{"key":"2_CR4","doi-asserted-by":"crossref","unstructured":"Tepperman, J., Narayanan, S.: Automatic syllable stress detection using prosodic features for pronunciation evaluation of language learners. In: Proceedings.(ICASSP 2005). IEEE International Conference on Acoustics, Speech, and Signal Processing, 2005, vol.\u00a01, pp. I\u2013937. IEEE (2005)","DOI":"10.1109\/ICASSP.2005.1415269"},{"issue":"2","key":"2_CR5","doi-asserted-by":"publisher","first-page":"989","DOI":"10.1121\/1.428279","volume":"107","author":"C Cucchiarini","year":"2000","unstructured":"Cucchiarini, C., Strik, H., Boves, L.: Quantitative assessment of second language learners\u2019 fluency by means of automatic speech recognition technology. J. Acoust. Soc. Am. 107(2), 989\u2013999 (2000)","journal-title":"J. Acoust. Soc. Am."},{"issue":"3","key":"2_CR6","doi-asserted-by":"publisher","first-page":"254","DOI":"10.1016\/j.specom.2009.11.001","volume":"52","author":"JP Arias","year":"2010","unstructured":"Arias, J.P., Yoma, N.B., Vivanco, H.: Automatic intonation assessment for computer aided language learning. Speech Commun. 52(3), 254\u2013267 (2010)","journal-title":"Speech Commun."},{"key":"2_CR7","doi-asserted-by":"crossref","unstructured":"Lin, B., Wang, L., Feng, X., Zhang, J.: Automatic scoring at multi-granularity for l2 pronunciation. In: Interspeech, pp. 3022\u20133026 (2020)","DOI":"10.21437\/Interspeech.2020-1282"},{"key":"2_CR8","doi-asserted-by":"crossref","unstructured":"Gong, Y., Chen, Z., Chu, I.H., Chang, P., Glass, J.: Transformer-based multi-aspect multi-granularity non-native English speaker pronunciation assessment. In: ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7262\u20137266. IEEE (2022)","DOI":"10.1109\/ICASSP43922.2022.9746743"},{"key":"2_CR9","doi-asserted-by":"publisher","first-page":"154","DOI":"10.1016\/j.specom.2014.12.008","volume":"67","author":"W Hu","year":"2015","unstructured":"Hu, W., Qian, Y., Soong, F.K., Wang, Y.: Improved mispronunciation detection with deep neural network trained acoustic models and transfer learning based logistic regression classifiers. Speech Commun. 67, 154\u2013166 (2015)","journal-title":"Speech Commun."},{"key":"2_CR10","doi-asserted-by":"crossref","unstructured":"Do, H., Kim, Y., Lee, G.G.: Hierarchical pronunciation assessment with multi-aspect attention. In: ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a01\u20135. IEEE (2023)","DOI":"10.1109\/ICASSP49357.2023.10095733"},{"key":"2_CR11","unstructured":"Peng, Y., Dalmia, S., Lane, I., Watanabe, S.: BranchFormer: parallel MLP-attention architectures to capture local and global context for speech recognition and understanding. In: International Conference on Machine Learning, pp. 17627\u201317643. PMLR (2022)"},{"key":"2_CR12","doi-asserted-by":"crossref","unstructured":"Zhang, J., et al.: speechocean762: an open-source non-native english speech corpus for pronunciation assessment. arXiv preprint arXiv:2104.01378 (2021)","DOI":"10.21437\/Interspeech.2021-1259"},{"key":"2_CR13","unstructured":"Soong, F.K., Lo, W.K., Nakamura, S.: Generalized word posterior probability (GWPP) for measuring reliability of recognized words. In: Proceedings of the SWIM2004, vol. 5 (2004)"},{"key":"2_CR14","doi-asserted-by":"crossref","unstructured":"Cheng, S., Liu, Z., Li, L., Tang, Z., Wang, D., Zheng, T.F.: ASR-free pronunciation assessment. arXiv preprint arXiv:2005.11902 (2020)","DOI":"10.21437\/Interspeech.2020-2623"},{"issue":"1","key":"2_CR15","doi-asserted-by":"publisher","first-page":"30","DOI":"10.1109\/TASL.2011.2134090","volume":"20","author":"GE Dahl","year":"2011","unstructured":"Dahl, G.E., Yu, D., Deng, L., Acero, A.: Context-dependent pre-trained deep neural networks for large-vocabulary speech recognition. IEEE Trans. Audio Speech Lang. Process. 20(1), 30\u201342 (2011)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"2_CR16","doi-asserted-by":"crossref","unstructured":"Lin, B., Wang, L.: Deep feature transfer learning for automatic pronunciation assessment. In: Interspeech, vol.\u00a02021, pp. 4438\u20134442 (2021)","DOI":"10.21437\/Interspeech.2021-931"},{"key":"2_CR17","doi-asserted-by":"crossref","unstructured":"Lin, B., Wang, L.: Attention-based multi-encoder automatic pronunciation assessment. In: ICASSP 2021-2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7743\u20137747. IEEE (2021)","DOI":"10.1109\/ICASSP39728.2021.9414451"},{"key":"2_CR18","doi-asserted-by":"crossref","unstructured":"Bann\u00f2, S., Matassoni, M.: Proficiency assessment of l2 spoken English using wav2vec 2.0. In: 2022 IEEE Spoken Language Technology Workshop (SLT), pp. 1088\u20131095. IEEE (2023)","DOI":"10.1109\/SLT54892.2023.10023019"},{"key":"2_CR19","doi-asserted-by":"crossref","unstructured":"Kim, E., Jeon, J.J., Seo, H., Kim, H.: Automatic pronunciation assessment using self-supervised speech representation learning. arXiv preprint arXiv:2204.03863 (2022)","DOI":"10.21437\/Interspeech.2022-10245"},{"key":"2_CR20","doi-asserted-by":"crossref","unstructured":"Chao, F.A., Lo, T.H., Wu, T.I., Sung, Y.T., Chen, B.: 3M: an effective multi-view, multi-granularity, and multi-aspect modeling approach to English pronunciation assessment. In: 2022 Asia-Pacific Signal and Information Processing Association Annual Summit and Conference (APSIPA ASC), pp. 575\u2013582. IEEE (2022)","DOI":"10.23919\/APSIPAASC55919.2022.9979979"},{"key":"2_CR21","doi-asserted-by":"crossref","unstructured":"Chao, F.A., Lo, T.H., Wu, T.I., Sung, Y.T., Chen, B.: A hierarchical context-aware modeling approach for multi-aspect and multi-granular pronunciation assessment. arXiv preprint arXiv:2305.18146 (2023)","DOI":"10.21437\/Interspeech.2023-550"},{"issue":"2\u20133","key":"2_CR22","doi-asserted-by":"publisher","first-page":"95","DOI":"10.1016\/S0167-6393(99)00044-8","volume":"30","author":"SM Witt","year":"2000","unstructured":"Witt, S.M., Young, S.J.: Phone-level pronunciation scoring and assessment for interactive language learning. Speech Commun. 30(2\u20133), 95\u2013108 (2000)","journal-title":"Speech Commun."},{"key":"2_CR23","doi-asserted-by":"crossref","unstructured":"Panayotov, V., Chen, G., Povey, D., Khudanpur, S.: LibriSpeech: an ASR corpus based on public domain audio books. In: 2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5206\u20135210. IEEE (2015)","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"2_CR24","doi-asserted-by":"crossref","unstructured":"Eyben, F., W\u00f6llmer, M., Schuller, B.: Opensmile: the Munich versatile and fast open-source audio feature extractor. In: Proceedings of the 18th ACM International Conference on Multimedia, pp. 1459\u20131462 (2010)","DOI":"10.1145\/1873951.1874246"},{"key":"2_CR25","doi-asserted-by":"crossref","unstructured":"Yang, Z., Yang, D., Dyer, C., He, X., Smola, A., Hovy, E.: Hierarchical attention networks for document classification. In: Proceedings of the 2016 conference of the North American chapter of the association for computational linguistics: human language technologies, pp. 1480\u20131489 (2016)","DOI":"10.18653\/v1\/N16-1174"},{"key":"2_CR26","doi-asserted-by":"crossref","unstructured":"Zhu, J., Zhang, C., Jurgens, D.: Phone-to-audio alignment without text: a semi-supervised approach. In: ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 8167\u20138171. IEEE (2022)","DOI":"10.1109\/ICASSP43922.2022.9746112"},{"key":"2_CR27","doi-asserted-by":"publisher","first-page":"554","DOI":"10.1109\/TASLP.2023.3335807","volume":"32","author":"HC Pei","year":"2023","unstructured":"Pei, H.C., Fang, H., Luo, X., Xu, X.S.: Gradformer: a framework for multi-aspect multi-granularity pronunciation assessment. IEEE\/ACM Trans. Audio Speech Lang. Process. 32, 554\u2013563 (2023)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."}],"container-title":["Lecture Notes in Computer Science","MultiMedia Modeling"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-96-2054-8_2","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,3,23]],"date-time":"2025-03-23T01:43:15Z","timestamp":1742694195000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-96-2054-8_2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9789819620531","9789819620548"],"references-count":27,"URL":"https:\/\/doi.org\/10.1007\/978-981-96-2054-8_2","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"3 January 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"MMM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Multimedia Modeling","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Nara","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Japan","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"9 January 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"11 January 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"31","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"mmm2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/mmm2025.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}