{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T15:48:24Z","timestamp":1784821704773,"version":"3.55.0"},"reference-count":28,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2024,6,1]],"date-time":"2024-06-01T00:00:00Z","timestamp":1717200000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,6,1]],"date-time":"2024-06-01T00:00:00Z","timestamp":1717200000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2024,6]]},"DOI":"10.1007\/s10772-024-10113-9","type":"journal-article","created":{"date-parts":[[2024,6,25]],"date-time":"2024-06-25T18:08:15Z","timestamp":1719338895000},"page":"413-424","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Anomaly detection with a variational autoencoder for Arabic mispronunciation detection"],"prefix":"10.1007","volume":"27","author":[{"given":"Meriem","family":"Lounis","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3834-2623","authenticated-orcid":false,"given":"Bilal","family":"Dendani","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1519-7075","authenticated-orcid":false,"given":"Halima","family":"Bahi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,6,25]]},"reference":[{"key":"10113_CR1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-47578-3","volume-title":"Outlier analysis","author":"CC Aggarwal","year":"2017","unstructured":"Aggarwal, C. C. (2017). Outlier analysis. Springer. https:\/\/doi.org\/10.1007\/978-3-319-47578-3"},{"issue":"7","key":"10113_CR2","doi-asserted-by":"publisher","first-page":"413","DOI":"10.3390\/info14070413","volume":"14","author":"A Ahmed","year":"2023","unstructured":"Ahmed, A., Bader, M., Shahin, I., Nassif, A. B., Werghi, N., & Basel, M. (2023). Arabic mispronunciation recognition system using LSTM network. Information, 14(7), 413. https:\/\/doi.org\/10.3390\/info14070413","journal-title":"Information"},{"issue":"6","key":"10113_CR3","doi-asserted-by":"publisher","first-page":"963","DOI":"10.3390\/electronics9060963","volume":"9","author":"S Akhtar","year":"2020","unstructured":"Akhtar, S., Hussain, F., Raja, F. R., Ehatisham-ul-haq, M., Baloch, N. K., Ishmanov, F., & Zikria, Y. B. (2020). Improving mispronunciation detection of Arabic words for non-native learners using deep convolutional neural network features. Electronics, 9(6), 963.","journal-title":"Electronics"},{"issue":"15","key":"10113_CR4","doi-asserted-by":"publisher","first-page":"2727","DOI":"10.3390\/math10152727","volume":"10","author":"M Algabri","year":"2022","unstructured":"Algabri, M., Mathkour, H., Alsulaiman, M., & Bencherif, M. A. (2022). Mispronunciation detection and diagnosis with articulatory-level feedback generation for non-native Arabic speech. Mathematics, 10(15), 2727. https:\/\/doi.org\/10.3390\/math10152727","journal-title":"Mathematics"},{"key":"10113_CR5","doi-asserted-by":"publisher","unstructured":"Aly, S. A., Salah, A., & Eraqi, H. M. (2021). ASMDD\u202f: Arabic speech mispronunciation detection dataset. arXiv preprint. https:\/\/doi.org\/10.48550\/arXiv.2111.01136","DOI":"10.48550\/arXiv.2111.01136"},{"issue":"1","key":"10113_CR6","first-page":"1","volume":"2","author":"J An","year":"2015","unstructured":"An, J., & Cho, S. (2015). Variational autoencoder based anomaly detection using reconstruction probability. Special Lecture on IE, 2(1), 1\u201318.","journal-title":"Special Lecture on IE"},{"issue":"1","key":"10113_CR7","doi-asserted-by":"publisher","first-page":"60","DOI":"10.4018\/IJCALLT.2020010105","volume":"10","author":"H Bahi","year":"2020","unstructured":"Bahi, H., & Necibi, K. (2020). Fuzzy logic applied for pronunciation assessment. International Journal of Computer-Assisted Language Learning and Teaching, 10(1), 60\u201372. https:\/\/doi.org\/10.4018\/IJCALLT.2020010105","journal-title":"International Journal of Computer-Assisted Language Learning and Teaching"},{"key":"10113_CR8","doi-asserted-by":"publisher","DOI":"10.1016\/j.apacoust.2023.109593","volume":"212","author":"SS Cal\u0131k","year":"2023","unstructured":"Cal\u0131k, S. S., Kucukmanisa, A., & Kilimci, Z. H. (2023). An ensemble-based framework for mispronunciation detection of Arabic phonemes. Applied Acoustics, 212, 109593.","journal-title":"Applied Acoustics"},{"key":"10113_CR9","unstructured":"Chalapathy, R., Menon, A. K., & Chawla, S. (2018). Anomaly detection using one-class neural networks. arXiv preprint. arXiv:1802.06360."},{"issue":"3","key":"10113_CR10","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/1541880.1541882","volume":"41","author":"V Chandola","year":"2009","unstructured":"Chandola, V., Banerjee, A., & Kumar, V. (2009). Anomaly detection: A survey. ACM Computing Surveys, 41(3), 1\u201358. https:\/\/doi.org\/10.1145\/1541880.1541882","journal-title":"ACM Computing Surveys"},{"key":"10113_CR11","doi-asserted-by":"publisher","unstructured":"Franco, H., Neumeyer, L., Kim, Y., & Ronen, O. (1997). Automatic pronunciation scoring for language instruction. In IEEE international conference on acoustics, speech, and signal processing (ICASSP) (pp. 1471\u20131474). https:\/\/doi.org\/10.1109\/ICASSP.1997.596227.","DOI":"10.1109\/ICASSP.1997.596227"},{"key":"10113_CR12","doi-asserted-by":"crossref","unstructured":"Frihia, H., & Bahi, H. (2020). One-class training for intrusion detection. In Proceedings of the 1st international conference on intelligent systems and pattern recognition (pp. 12\u201316).","DOI":"10.1145\/3432867.3432898"},{"key":"10113_CR13","doi-asserted-by":"crossref","unstructured":"Hu, W., Qian, Y., & Soong, F. K. (2014). A new neural network based logistic regression classifier for improving mispronunciation detection of L2 language learners. In The 9th international symposium on Chinese spoken language processing (pp. 245\u2013249).","DOI":"10.1109\/ISCSLP.2014.6936712"},{"key":"10113_CR14","doi-asserted-by":"publisher","unstructured":"Kingma, D. P., & Welling, M. (2013). Auto-encoding variational bayes. arXiv preprint. https:\/\/doi.org\/10.48550\/arXiv.1312.6114.","DOI":"10.48550\/arXiv.1312.6114"},{"issue":"4","key":"10113_CR15","doi-asserted-by":"publisher","first-page":"307","DOI":"10.1561\/2200000056","volume":"12","author":"DP Kingma","year":"2019","unstructured":"Kingma, D. P., & Welling, M. (2019). An introduction to variational autoencoders. Foundations and Trends in Machine Learning, 12(4), 307\u2013392. https:\/\/doi.org\/10.1561\/2200000056","journal-title":"Foundations and Trends in Machine Learning"},{"key":"10113_CR16","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-023-17899-x","author":"M Lounis","year":"2024","unstructured":"Lounis, M., Dendani, B., & Bahi, H. (2024). Mispronunciation detection and diagnosis using deep neural networks: A systematic review. Multimedia Tools and Application. https:\/\/doi.org\/10.1007\/s11042-023-17899-x","journal-title":"Multimedia Tools and Application"},{"key":"10113_CR17","doi-asserted-by":"publisher","unstructured":"Masuyama, Y., Yatabe, K., Koizumi, Y., Oikawa, Y., & Harada, N. (2019). Deep Griffin-Lim iteration. In IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 61\u201365). https:\/\/doi.org\/10.1109\/icassp.2019.8682744","DOI":"10.1109\/icassp.2019.8682744"},{"key":"10113_CR18","doi-asserted-by":"publisher","first-page":"52589","DOI":"10.1109\/ACCESS.2019.2912648","volume":"7","author":"F Nazir","year":"2019","unstructured":"Nazir, F., Majeed, M. N., Ghazanfar, M. A., & Maqsood, M. (2019). Mispronunciation detection using deep convolutional neural network features and transfer learning-based model for Arabic phonemes. IEEE Access, 7, 52589\u201352608. https:\/\/doi.org\/10.1109\/ACCESS.2019.2912648","journal-title":"IEEE Access"},{"issue":"1","key":"10113_CR19","doi-asserted-by":"publisher","first-page":"37","DOI":"10.1007\/s10772-014-9248-2","volume":"18","author":"K Necibi","year":"2015","unstructured":"Necibi, K., & Bahi, H. (2015). A statistical-based decision for Arabic pronunciation assessment. International Journal of Speech Technology, 18(1), 37\u201344. https:\/\/doi.org\/10.1007\/s10772-014-9248-2","journal-title":"International Journal of Speech Technology"},{"key":"10113_CR20","unstructured":"Odaibo, S. (2019). Tutorial: Deriving the standard variational autoencoder (VAE) loss function. arXiv preprint. http:\/\/arxiv.org\/abs\/1907.08956"},{"key":"10113_CR21","unstructured":"Ruff, L., Vandermeulen, R., Goernitz, N., Deecke, L., Siddiqui, S. A., Binder, A., M\u00fcller, E., & Kloft, M. (2018). Deep one-class classification. In Proceedings of the 35th international conference on machine learning (PMLR) (pp. 4393\u20134402). https:\/\/proceedings.mlr.press\/v80\/ruff18a.html."},{"key":"10113_CR22","doi-asserted-by":"publisher","unstructured":"Samir, A., Abdou, S. M., Khalil, A. H., & Rashwan, M. (2007). Enhancing usability of CAPL system for Qur\u2019an recitation learning. In Interspeech (pp. 214\u2013217). https:\/\/doi.org\/10.21437\/interspeech.2007-89","DOI":"10.21437\/interspeech.2007-89"},{"key":"10113_CR23","doi-asserted-by":"publisher","first-page":"29","DOI":"10.1016\/j.specom.2019.06.003","volume":"111","author":"M Shahin","year":"2019","unstructured":"Shahin, M., & Ahmed, B. (2019). Anomaly detection based pronunciation verification approach using speech attribute features. Speech Communication, 111, 29\u201343. https:\/\/doi.org\/10.1016\/j.specom.2019.06.003","journal-title":"Speech Communication"},{"issue":"2","key":"10113_CR24","doi-asserted-by":"publisher","first-page":"400","DOI":"10.1109\/JSTSP.2019.2959393","volume":"14","author":"M Shahin","year":"2020","unstructured":"Shahin, M., Zafar, U., & Ahmed, B. (2020). The automatic detection of speech disorders in children: Challenges, opportunities, and preliminary results. IEEE Journal of Selected Topics in Signal Processing, 14(2), 400\u2013412. https:\/\/doi.org\/10.1109\/JSTSP.2019.2959393","journal-title":"IEEE Journal of Selected Topics in Signal Processing"},{"key":"10113_CR25","unstructured":"Tschannen, M., Bachem, O., & Lucic, M. (2018). Recent advances in autoencoder-based representation learning. In Third workshop on Bayesian deep learning (NeurIPS 2018)."},{"key":"10113_CR26","doi-asserted-by":"publisher","first-page":"153651","DOI":"10.1109\/ACCESS.2020.3018151","volume":"8","author":"R Wei","year":"2020","unstructured":"Wei, R., Garcia, C., El-Sayed, A., Peterson, V., & Mahmood, A. (2020). Variations in variational autoencoders\u2014a comparative evaluation. IEEE Access, 8, 153651\u2013153670. https:\/\/doi.org\/10.1109\/ACCESS.2020.3018151","journal-title":"IEEE Access"},{"issue":"2\u20133","key":"10113_CR27","doi-asserted-by":"publisher","first-page":"95","DOI":"10.1016\/S0167-6393(99)00044-8","volume":"30","author":"SM Witt","year":"2000","unstructured":"Witt, S. M., & Young, S. J. (2000). Phone-level pronunciation scoring and assessment for interactive language learning. Speech Communication, 30(2\u20133), 95\u2013108. https:\/\/doi.org\/10.1016\/S0167-6393(99)00044-8","journal-title":"Speech Communication"},{"issue":"2","key":"10113_CR28","first-page":"49","volume":"28","author":"L Yang","year":"2019","unstructured":"Yang, L., Xie, Y., & Zhang, J. (2019). Pronunciation erroneous tendency detection with combination of convolutional neural network and long short-term memory. International Journal of Asian Language Processing, 28(2), 49\u201366.","journal-title":"International Journal of Asian Language Processing"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10113-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10772-024-10113-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10113-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,22]],"date-time":"2024-07-22T16:07:56Z","timestamp":1721664476000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10772-024-10113-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,6]]},"references-count":28,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2024,6]]}},"alternative-id":["10113"],"URL":"https:\/\/doi.org\/10.1007\/s10772-024-10113-9","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"value":"1381-2416","type":"print"},{"value":"1572-8110","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,6]]},"assertion":[{"value":"23 January 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 May 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 June 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no competing interests to declare that are relevant to the article\u2019s content.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}]}}