{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T23:45:45Z","timestamp":1782949545648,"version":"3.54.5"},"publisher-location":"Cham","reference-count":30,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031936968","type":"print"},{"value":"9783031936975","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,7,24]],"date-time":"2025-07-24T00:00:00Z","timestamp":1753315200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,7,24]],"date-time":"2025-07-24T00:00:00Z","timestamp":1753315200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-031-93697-5_14","type":"book-chapter","created":{"date-parts":[[2025,7,23]],"date-time":"2025-07-23T13:48:06Z","timestamp":1753278486000},"page":"190-205","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["VLRASN-112: A Visual Lip Reading Dataset for\u00a0Assamese Compound Numeric Sequence Recognition"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-6227-7836","authenticated-orcid":false,"given":"Meghali","family":"Deka","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-6658-7436","authenticated-orcid":false,"given":"Vaibhav","family":"Gavit","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2885-0026","authenticated-orcid":false,"given":"Prithwijit","family":"Guha","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5869-1057","authenticated-orcid":false,"given":"Sukumar","family":"Nandi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9051-1255","authenticated-orcid":false,"given":"Priyankoo","family":"Sarmah","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,7,24]]},"reference":[{"key":"14_CR1","unstructured":"Aggarwal, Y., Guha, P.: FDLite: a single stage lightweight face detector network. arXiv preprint arXiv:2406.19107 (2024)"},{"key":"14_CR2","doi-asserted-by":"crossref","unstructured":"Anina, I., Zhou, Z., Zhao, G., Pietik\u00e4inen, M.: OuluVS2: a multi-view audiovisual database for non-rigid mouth motion analysis. In: 2015 11th IEEE International Conference and Workshops on Automatic Face and Gesture Recognition (FG), vol.\u00a01, pp.\u00a01\u20135. IEEE (2015)","DOI":"10.1109\/FG.2015.7163155"},{"key":"14_CR3","unstructured":"Antar, S., Sagheer, A.: Audio visual Arabic speech (AVAS) database for human-computer interaction applications. Int. J. Adv. Res. Comput. Sci. Softw. Eng. 3(9) (2013)"},{"key":"14_CR4","unstructured":"Assael, Y.M., Shillingford, B., Whiteson, S., De\u00a0Freitas, N.: LIPNet: end-to-end sentence-level lipreading. arXiv preprint arXiv:1611.01599 (2016)"},{"key":"14_CR5","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"625","DOI":"10.1007\/3-540-44887-X_74","volume-title":"Audio- and Video-Based Biometric Person Authentication","author":"E Bailly-Bailli\u00e9re","year":"2003","unstructured":"Bailly-Bailli\u00e9re, E., et al.: The BANCA database and evaluation protocol. In: Kittler, J., Nixon, M.S. (eds.) AVBPA 2003. LNCS, vol. 2688, pp. 625\u2013638. Springer, Heidelberg (2003). https:\/\/doi.org\/10.1007\/3-540-44887-X_74"},{"key":"14_CR6","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"259","DOI":"10.1007\/978-3-642-15760-8_33","volume-title":"Text, Speech and Dialogue","author":"AG Chitu","year":"2010","unstructured":"Chitu, A.G., Driel, K., Rothkrantz, L.: Automatic Lip reading in the Dutch language using active appearance models on high speed recordings. In: Sojka, P., Hor\u00e1k, A., Kope\u010dek, I., Pala, K. (eds.) TSD 2010. LNCS (LNAI), vol. 6231, pp. 259\u2013266. Springer, Heidelberg (2010). https:\/\/doi.org\/10.1007\/978-3-642-15760-8_33"},{"key":"14_CR7","unstructured":"Choudhury, P., Guha, P., Nandi, S.: Image caption synthesis for low resource assamese language using bi-LSTM with bilinear attention. In: Proceedings of the 37th Pacific Asia Conference on Language, Information and Computation, pp. 743\u2013752 (2023)"},{"key":"14_CR8","unstructured":"Estellers, V., Thiran, J.P.: Multipose audio-visual speech recognition. In: 2011 19th European Signal Processing Conference, pp. 1065\u20131069. IEEE (2011)"},{"key":"14_CR9","doi-asserted-by":"crossref","unstructured":"Estival, D., Cassidy, S., Cox, F., Burnham, D.: Austalk: an audio-visual corpus of Australian English (2014)","DOI":"10.63317\/5fmuqesg3xoi"},{"issue":"1","key":"14_CR10","doi-asserted-by":"publisher","first-page":"147","DOI":"10.17576\/jkukm-2024-36(1)-14","volume":"36","author":"H Fazli\u0107a","year":"2024","unstructured":"Fazli\u0107a, H., Abd Almisrea, A., Tahirb, N.M.: Deep learning-based audio-visual speech recognition for Bosnian digits. Jurnal Kejuruteraan 36(1), 147\u2013154 (2024)","journal-title":"Jurnal Kejuruteraan"},{"key":"14_CR11","doi-asserted-by":"publisher","first-page":"2076","DOI":"10.1109\/TASLP.2022.3182274","volume":"30","author":"A Fernandez-Lopez","year":"2022","unstructured":"Fernandez-Lopez, A., Sukno, F.M.: End-to-end lip-reading without large-scale data. IEEE\/ACM Trans. Audio, Speech Lang. Process. 30, 2076\u20132090 (2022)","journal-title":"IEEE\/ACM Trans. Audio, Speech Lang. Process."},{"key":"14_CR12","unstructured":"General, R.: Census commissioner, India. Census India 2000 (2011)"},{"key":"14_CR13","doi-asserted-by":"crossref","unstructured":"Graves, A., Fern\u00e1ndez, S., Gomez, F., Schmidhuber, J.: Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks. In: Proceedings of the 23rd International Conference on Machine Learning, pp. 369\u2013376 (2006)","DOI":"10.1145\/1143844.1143891"},{"issue":"2B","key":"14_CR14","first-page":"163","volume":"33","author":"M Igras","year":"2012","unstructured":"Igras, M., Zi\u00f3\u0142ko, B., Jadczyk, T.: Audiovisual database of polish speech recordings. Studia Informatica 33(2B), 163\u2013172 (2012)","journal-title":"Studia Informatica"},{"key":"14_CR15","doi-asserted-by":"crossref","unstructured":"Imamura, A., Arizumi, N.: Gabor filter incorporated CNN for compression. In: 2021 36th International Conference on Image and Vision Computing New Zealand (IVCNZ), pp.\u00a01\u20135. IEEE (2021)","DOI":"10.1109\/IVCNZ54163.2021.9653342"},{"key":"14_CR16","first-page":"1755","volume":"10","author":"DE King","year":"2009","unstructured":"King, D.E.: DLIB-ML: a machine learning toolkit. J. Mach. Learn. Res. 10, 1755\u20131758 (2009)","journal-title":"J. Mach. Learn. Res."},{"key":"14_CR17","doi-asserted-by":"crossref","unstructured":"Lee, B., et al.: AVICAR: audio-visual speech corpus in a car environment. In: Interspeech, pp. 2489\u20132492 (2004)","DOI":"10.21437\/Interspeech.2004-424"},{"issue":"8","key":"14_CR18","doi-asserted-by":"publisher","first-page":"1599","DOI":"10.3390\/app9081599","volume":"9","author":"Y Lu","year":"2019","unstructured":"Lu, Y., Li, H.: Automatic lip-reading system based on deep convolutional neural network and attention-based long short-term memory. Appl. Sci. 9(8), 1599 (2019)","journal-title":"Appl. Sci."},{"key":"14_CR19","unstructured":"Lucey, P., Potamianos, G., Sridharan, S.: Patch-based analysis of visual speech from multiple views. In: Proceedings of the International Conference on Auditory-Visual Speech Processing 2008, pp. 69\u201374. AVISA (2008)"},{"key":"14_CR20","unstructured":"Messer, K., Matas, J., Kittler, J., Luettin, J., Maitre, G., et\u00a0al.: XM2VTSDB: the extended m2vts database. In: Second International Conference on Audio and Video-based Biometric Person Authentication, vol.\u00a0964, pp. 965\u2013966. Citeseer (1999)"},{"key":"14_CR21","unstructured":"Movellan, J.: Visual speech recognition with stochastic networks. Adv. Neural Info. Process. Syst. 7 (1994)"},{"key":"14_CR22","doi-asserted-by":"crossref","unstructured":"Ortega, A., et al.: Av@ car: a Spanish multichannel multimodal corpus for in-vehicle automatic audio-visual speech recognition. In: LREC (2004)","DOI":"10.63317\/26hq9jzng47p"},{"key":"14_CR23","doi-asserted-by":"crossref","unstructured":"Patterson, E.K., Gurbuz, S., Tufekci, Z., Gowdy, J.N.: Cuave: a new audio-visual database for multimodal human-computer interface research. In: 2002 IEEE International conference on acoustics, speech, and signal processing, vol.\u00a02, pp. II\u20132017. IEEE (2002)","DOI":"10.1109\/ICASSP.2002.5745028"},{"key":"14_CR24","unstructured":"Petr, C., Milo\u0161, \u017d., Zden\u011bk, K., Jakub, K., Jan, Z., Lud\u011bk, M.: Design and recording of czech speech corpus for audio-visual continuous speech recognition (2005)"},{"key":"14_CR25","doi-asserted-by":"crossref","unstructured":"Petridis, S., Shen, J., Cetin, D., Pantic, M.: Visual-only recognition of normal, whispered and silent speech. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6219\u20136223. IEEE (2018)","DOI":"10.1109\/ICASSP.2018.8461596"},{"key":"14_CR26","first-page":"480","volume":"720","author":"S Tamura","year":"2010","unstructured":"Tamura, S., et al.: Censrec-1-AV: an audio-visual corpus for noisy bimodal speech recognition. Training 720, 480 (2010)","journal-title":"Training"},{"key":"14_CR27","doi-asserted-by":"crossref","unstructured":"Vignesh, P., Shrihari, A., Guha, P.: EuWod-16: an extended dataset for underwater object detection. In: International Conference on Computer Vision and Image Processing, pp. 434\u2013445. Springer (2023)","DOI":"10.1007\/978-3-031-58535-7_36"},{"key":"14_CR28","doi-asserted-by":"crossref","unstructured":"Wang, H., Cui, B., Yuan, Q., Pu, G., Liu, X., Zhu, J.: Mini-3DCVT: a lightweight lip-reading method based on 3D convolution visual transformer. Vis. Comput. 41(3), 1\u201313 (2024)","DOI":"10.1007\/s00371-024-03515-y"},{"key":"14_CR29","doi-asserted-by":"crossref","unstructured":"Xu, K., Li, D., Cassimatis, N., Wang, X.: LCANet: end-to-end lipreading with cascaded attention-CTC. In: 2018 13th IEEE International Conference on Automatic Face & Gesture Recognition (FG 2018), pp. 548\u2013555. IEEE (2018)","DOI":"10.1109\/FG.2018.00088"},{"issue":"1s","key":"14_CR30","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3524620","volume":"19","author":"F Xue","year":"2023","unstructured":"Xue, F., et al.: LCSNet: end-to-end lipreading with channel-aware feature selection. ACM Trans. Multimed. Comput. Commun. Appl. 19(1s), 1\u201321 (2023)","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."}],"container-title":["Communications in Computer and Information Science","Computer Vision and Image Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-93697-5_14","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T23:23:55Z","timestamp":1782948235000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-93697-5_14"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,24]]},"ISBN":["9783031936968","9783031936975"],"references-count":30,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-93697-5_14","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"value":"1865-0929","type":"print"},{"value":"1865-0937","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,7,24]]},"assertion":[{"value":"24 July 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"CVIP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Computer Vision and Image Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Chennai","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"India","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 December 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 December 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"9","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"cvip2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/cvip2024.iiitdm.ac.in\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}