{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T19:05:34Z","timestamp":1761419134895,"version":"build-2065373602"},"reference-count":34,"publisher":"Springer Science and Business Media LLC","issue":"14","license":[{"start":{"date-parts":[[2025,9,22]],"date-time":"2025-09-22T00:00:00Z","timestamp":1758499200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,9,22]],"date-time":"2025-09-22T00:00:00Z","timestamp":1758499200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"the Key Project of Science and Technology Research Program of Chongqing Municipal Education Commission","award":["KJZD-K202301505"],"award-info":[{"award-number":["KJZD-K202301505"]}]},{"name":"the General Program of Chongqing Science and Technology Commission","award":["CSTC2021jcyj-msxm3332"],"award-info":[{"award-number":["CSTC2021jcyj-msxm3332"]}]},{"name":"the Scientific and Technological Research Program of Chongqing Municipal Education Commission","award":["KJQN202301502"],"award-info":[{"award-number":["KJQN202301502"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SIViP"],"published-print":{"date-parts":[[2025,12]]},"DOI":"10.1007\/s11760-025-04788-z","type":"journal-article","created":{"date-parts":[[2025,9,22]],"date-time":"2025-09-22T13:12:52Z","timestamp":1758546772000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Audio visual chinese speech recognition in daily scene based on cross modal context encoder"],"prefix":"10.1007","volume":"19","author":[{"given":"Yijun","family":"Liu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhihua","family":"Qu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jie","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qian","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bowen","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wujun","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,9,22]]},"reference":[{"key":"4788_CR1","first-page":"1","volume":"1","author":"H Ahlawat","year":"2025","unstructured":"Ahlawat, H., Aggarwal, N., Gupta, D.: Automatic speech recognition: a survey of deep learning techniques and approaches. Int J Cogn Comput Eng 1, 1 (2025)","journal-title":"Int J Cogn Comput Eng"},{"key":"4788_CR2","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2024.102422","volume":"109","author":"H Kheddar","year":"2024","unstructured":"Kheddar, H., Hemis, M., Himeur, Y.: Automatic speech recognition using advanced deep learning approaches: a survey. Inf. Fusion 109, 102422 (2024)","journal-title":"Inf. Fusion"},{"issue":"1","key":"4788_CR3","doi-asserted-by":"publisher","first-page":"2363","DOI":"10.1038\/s41467-025-57629-5","volume":"16","author":"T Liu","year":"2025","unstructured":"Liu, T., Zhang, M., Li, Z., Dou, H., Zhang, W., Yang, J., Wu, P., Li, D., Mu, X.: Machine learning-assisted wearable sensing systems for speech recognition and interaction. Nat. Commun. 16(1), 2363 (2025)","journal-title":"Nat. Commun."},{"key":"4788_CR4","unstructured":"He, X., Whitehill, J.: Survey of end-to-end multi-speaker automatic speech recognition for monaural audio. arXiv preprint arXiv:2505.10975 (2025)"},{"key":"4788_CR5","doi-asserted-by":"crossref","unstructured":"Parisae, V., Bhavanam, S.N., Devi, M.V.: Progressive learning framework for speech enhancement using multi-scale convolution and s-tcn. In: 2024 8th International Conference on Inventive Systems and Control (ICISC), pp. 83\u201389 (2024). IEEE","DOI":"10.1109\/ICISC62624.2024.00021"},{"key":"4788_CR6","doi-asserted-by":"crossref","unstructured":"Parisae, V., Nagakishore\u00a0Bhavanam, S.: Stacked u-net with time\u2013frequency attention and deep connection net for single channel speech enhancement. Int. J. Image Graph, 2550067 (2024)","DOI":"10.1142\/S0219467825500676"},{"issue":"4","key":"4788_CR7","first-page":"10907","volume":"46","author":"V Parisae","year":"2024","unstructured":"Parisae, V., Nagakishore Bhavanam, S.: Multi scale encoder-decoder network with time frequency attention and s-tcn for single channel speech enhancement. J. Intell. Fuzzy Syst 46(4), 10907\u201310907 (2024)","journal-title":"J. Intell. Fuzzy Syst"},{"issue":"01","key":"4788_CR8","doi-asserted-by":"publisher","first-page":"2550001","DOI":"10.1142\/S0219467825500019","volume":"25","author":"C Jannu","year":"2025","unstructured":"Jannu, C., Vanambathina, S.D.: An overview of speech enhancement based on deep learning techniques. Int. J. Image Graph 25(01), 2550001 (2025)","journal-title":"Int. J. Image Graph"},{"issue":"1","key":"4788_CR9","doi-asserted-by":"publisher","first-page":"70016","DOI":"10.1111\/coin.70016","volume":"41","author":"C Jannu","year":"2025","unstructured":"Jannu, C., Burra, M., Vanambathina, S.D., Parisae, V.: Real-time single channel speech enhancement using triple attention and stacked squeeze-tcn. Comput. Intell. 41(1), 70016 (2025)","journal-title":"Comput. Intell."},{"key":"4788_CR10","doi-asserted-by":"crossref","unstructured":"Sheng, C., Kuang, G., Bai, L., Hou, C., Guo, Y., Xu, X., Pietik\u00e4inen, M., Liu, L.: Deep learning for visual speech analysis: A survey. IEEE Transactions on Pattern Analysis and Machine Intelligence (2024)","DOI":"10.1109\/TPAMI.2024.3376710"},{"key":"4788_CR11","first-page":"9211","volume":"33","author":"X Zhang","year":"2019","unstructured":"Zhang, X., Gong, H., Dai, X., Yang, F., Liu, N., Liu, M.: Understanding pictograph with facial features: end-to-end sentence-level lip reading of chinese. Proc. AAAI Conf. Artif. Intell 33, 9211\u20139218 (2019)","journal-title":"Proc. AAAI Conf. Artif. Intell"},{"key":"4788_CR12","doi-asserted-by":"crossref","unstructured":"Chen, C., Liu, Z., Li, X., Li, L., Wang, D.: Cnvsrc 2023: The first chinese continuous visual speech recognition challenge. arXiv preprint arXiv:2406.10313 (2024)","DOI":"10.21437\/Interspeech.2024-2509"},{"key":"4788_CR13","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Xu, R., Song, M.: A cascade sequence-to-sequence model for chinese mandarin lip reading. In: Proceedings of the 1st ACM International Conference on Multimedia in Asia, pp. 1\u20136 (2019)","DOI":"10.1145\/3338533.3366579"},{"key":"4788_CR14","unstructured":"Narayan, S., Djilali, Y.A.D., Singh, A., Bihan, E.L., Hacid, H.: Visper: Multilingual audio-visual speech recognition. arXiv preprint arXiv:2406.00038 (2024)"},{"key":"4788_CR15","doi-asserted-by":"publisher","DOI":"10.1016\/j.image.2023.117002","volume":"117","author":"B Sun","year":"2023","unstructured":"Sun, B., Xie, D., Shi, H.: Malip: modal amplification lipreading based on reconstructed audio features. Signal Process. Image Commun 117, 117002 (2023)","journal-title":"Signal Process. Image Commun"},{"key":"4788_CR16","doi-asserted-by":"crossref","unstructured":"Chen, H., Zhou, H., Du, J., Lee, C.-H., Chen, J., Watanabe, S., Siniscalchi, S.M., Scharenborg, O., Liu, D.-Y., Yin, B.-C., etal: The first multimodal information based speech processing (misp) challenge: Data, tasks, baselines and results. In: ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 9266\u20139270 (2022). IEEE","DOI":"10.1109\/ICASSP43922.2022.9746683"},{"key":"4788_CR17","doi-asserted-by":"crossref","unstructured":"Sheng, C., Pietik\u00e4inen, M., Tian, Q., Liu, L.: Cross-modal self-supervised learning for lip reading: When contrastive learning meets adversarial training. In: Proceedings of the 29th ACM International Conference on Multimedia, pp. 2456\u20132464 (2021)","DOI":"10.1145\/3474085.3475415"},{"key":"4788_CR18","doi-asserted-by":"crossref","unstructured":"Pan, X., Chen, P., Gong, Y., Zhou, H., Wang, X., Lin, Z.: Leveraging unimodal self-supervised learning for multimodal audio-visual speech recognition. arXiv preprint arXiv:2203.07996 (2022)","DOI":"10.18653\/v1\/2022.acl-long.308"},{"key":"4788_CR19","doi-asserted-by":"crossref","unstructured":"Li, J., Li, C., Wu, Y., Qian, Y.: Robust audio-visual asr with unified cross-modal attention. In: ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1\u20135 (2023). IEEE","DOI":"10.1109\/ICASSP49357.2023.10096893"},{"key":"4788_CR20","doi-asserted-by":"crossref","unstructured":"Gimeno-G\u00f3mez, D., Mart\u00ednez-Hinarejos, C.-D.: Tailored design of audio-visual speech recognition models using branchformers. arXiv preprint arXiv:2407.06606 (2024)","DOI":"10.1016\/j.csl.2025.101811"},{"key":"4788_CR21","doi-asserted-by":"crossref","unstructured":"Afouras, T., Chung, J.S., Zisserman, A.: Deep lip reading: a comparison of models and an online application. arXiv preprint arXiv:1806.06053 (2018)","DOI":"10.21437\/Interspeech.2018-1943"},{"key":"4788_CR22","doi-asserted-by":"crossref","unstructured":"Ma, P., Haliassos, A., Fernandez-Lopez, A., Chen, H., Petridis, S., Pantic, M.: Auto-avsr: Audio-visual speech recognition with automatic labels. In: ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1\u20135 (2023). IEEE","DOI":"10.1109\/ICASSP49357.2023.10096889"},{"key":"4788_CR23","doi-asserted-by":"crossref","unstructured":"Graves, A., Fern\u00e1ndez, S., Gomez, F., Schmidhuber, J.: Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks. In: Proceedings of the 23rd International Conference on Machine Learning, pp. 369\u2013376 (2006)","DOI":"10.1145\/1143844.1143891"},{"key":"4788_CR24","doi-asserted-by":"crossref","unstructured":"Graves, A.: Sequence transduction with recurrent neural networks. arXiv preprint arXiv:1211.3711 (2012)","DOI":"10.1007\/978-3-642-24797-2"},{"key":"4788_CR25","doi-asserted-by":"crossref","unstructured":"Burchi, M., Puvvada, K.C., Balam, J., Ginsburg, B., Timofte, R.: Multilingual audio-visual speech recognition with hybrid ctc\/rnn-t fast conformer. In: ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 10211\u201310215 (2024). IEEE","DOI":"10.1109\/ICASSP48485.2024.10445891"},{"key":"4788_CR26","unstructured":"Peng, Y., Dalmia, S., Lane, I., Watanabe, S.: Branchformer: Parallel mlp-attention architectures to capture local and global context for speech recognition and understanding. In: International Conference on Machine Learning, pp. 17627\u201317643 (2022). PMLR"},{"key":"4788_CR27","doi-asserted-by":"crossref","unstructured":"Wang, Y., Chen, Z., Zheng, C., Zhang, Y., Han, W., Haghani, P.: Accelerating rnn-t training and inference using ctc guidance. In: ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1\u20135 (2023). IEEE","DOI":"10.1109\/ICASSP49357.2023.10096065"},{"key":"4788_CR28","doi-asserted-by":"crossref","unstructured":"Qi, D., Tan, W., Yao, Q., Liu, J.: Yolo5face: Why reinventing a face detector. In: European Conference on Computer Vision, pp. 228\u2013244 (2022). Springer","DOI":"10.1007\/978-3-031-25072-9_15"},{"key":"4788_CR29","doi-asserted-by":"crossref","unstructured":"Gulati, A., Qin, J., Chiu, C.-C., Parmar, N., Zhang, Y., Yu, J., Han, W., Wang, S., Zhang, Z., Wu, Y., et al.: Conformer: Convolution-augmented transformer for speech recognition. arXiv preprint arXiv:2005.08100 (2020)","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"4788_CR30","doi-asserted-by":"crossref","unstructured":"Ren, X., Li, C., Wang, S., Li, B.: Practice of the conformer enhanced audio-visual hubert on mandarin and english. In: ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1\u20135 (2023). IEEE","DOI":"10.1109\/ICASSP49357.2023.10094579"},{"key":"4788_CR31","unstructured":"Chung, J., Zisserman, A.: Lip reading in profile. In: British Machine Vision Conference, 2017. British Machine Vision Association and Society for Pattern Recognition"},{"key":"4788_CR32","first-page":"6917","volume":"34","author":"Y Zhao","year":"2020","unstructured":"Zhao, Y., Xu, R., Wang, X., Hou, P., Tang, H., Song, M.: Hearing lips: improving lip reading by distilling speech recognizers. Proc. AAAI Conf. Artif. Intell 34, 6917\u20136924 (2020)","journal-title":"Proc. AAAI Conf. Artif. Intell"},{"issue":"2","key":"4788_CR33","doi-asserted-by":"publisher","first-page":"2772","DOI":"10.1109\/TNNLS.2022.3191677","volume":"35","author":"L Qu","year":"2022","unstructured":"Qu, L., Weber, C., Wermter, S.: Lipsound2: self-supervised pre-training for lip-to-speech reconstruction and lip reading. IEEE Trans Neural Netw Learn Syst 35(2), 2772\u20132782 (2022)","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"4788_CR34","unstructured":"Djilali, Y.A.D., Narayan, S., LeBihan, E., Boussaid, H., Almazrouei, E., Debbah, M.: Do vsr models generalize beyond lrs3? In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 6635\u20136644 (2024)"}],"container-title":["Signal, Image and Video Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-025-04788-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11760-025-04788-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-025-04788-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T18:57:11Z","timestamp":1761418631000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11760-025-04788-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,22]]},"references-count":34,"journal-issue":{"issue":"14","published-print":{"date-parts":[[2025,12]]}},"alternative-id":["4788"],"URL":"https:\/\/doi.org\/10.1007\/s11760-025-04788-z","relation":{},"ISSN":["1863-1703","1863-1711"],"issn-type":[{"type":"print","value":"1863-1703"},{"type":"electronic","value":"1863-1711"}],"subject":[],"published":{"date-parts":[[2025,9,22]]},"assertion":[{"value":"30 May 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 August 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 September 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 September 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"1204"}}