{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T23:47:13Z","timestamp":1782949633889,"version":"3.54.5"},"publisher-location":"Cham","reference-count":25,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031936968","type":"print"},{"value":"9783031936975","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,7,24]],"date-time":"2025-07-24T00:00:00Z","timestamp":1753315200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,7,24]],"date-time":"2025-07-24T00:00:00Z","timestamp":1753315200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-031-93697-5_12","type":"book-chapter","created":{"date-parts":[[2025,7,23]],"date-time":"2025-07-23T13:48:52Z","timestamp":1753278532000},"page":"159-174","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["TDIUC-AVQA: A Visual Question Answering Dataset in\u00a0Low-Resource Assamese Language"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7188-0111","authenticated-orcid":false,"given":"Nazreena","family":"Rahman","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-1159-3118","authenticated-orcid":false,"given":"Pankaj","family":"Choudhury","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2885-0026","authenticated-orcid":false,"given":"Prithwijit","family":"Guha","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0024-3358","authenticated-orcid":false,"given":"Ashish","family":"Anand","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5869-1057","authenticated-orcid":false,"given":"Sukumar","family":"Nandi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,7,24]]},"reference":[{"key":"12_CR1","doi-asserted-by":"crossref","unstructured":"Anderson, P., et al.: Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6077\u20136086 (2018)","DOI":"10.1109\/CVPR.2018.00636"},{"key":"12_CR2","unstructured":"Anh, V.D., Minh, P.Q.N., Tran, G.S.: A novel pretrained general-purpose vision language model for the Vietnamese language. In: ACM Transactions on Asian and Low-Resource Language Information Processing (2024)"},{"key":"12_CR3","doi-asserted-by":"crossref","unstructured":"Antol, S., et al.: VQA: visual question answering. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2425\u20132433 (2015)","DOI":"10.1109\/ICCV.2015.279"},{"key":"12_CR4","unstructured":"Bahdanau, D., Cho, K., Bengio, Y.: Neural machine translation by jointly learning to align and translate. arXiv preprint arXiv:1409.0473 (2014)"},{"key":"12_CR5","doi-asserted-by":"crossref","unstructured":"Chandrasekar, A., Shimpi, A., Naik, D.: Indic visual question answering. In: 2022 IEEE International Conference on Signal Processing and Communications (SPCOM), pp.\u00a01\u20135. IEEE (2022)","DOI":"10.1109\/SPCOM55316.2022.9840835"},{"key":"12_CR6","unstructured":"Choudhury, P., Guha, P., Nandi, S.: Image caption synthesis for low resource Assamese language using bi-LSTM with bilinear attention. In: Proceedings of the 37th Pacific Asia Conference on Language, Information and Computation, pp. 743\u2013752 (2023)"},{"key":"12_CR7","doi-asserted-by":"crossref","unstructured":"Choudhury, P., Guha, P., Nandi, S.: Impact of language-specific training on image caption synthesis: a case study on low-resource Assamese language. Int. J. Asian Lang. Process. (2024)","DOI":"10.1142\/S2717554524500048"},{"key":"12_CR8","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: ImageNet: a large-scale hierarchical image database. In: 2009 IEEE Conference on Computer Vision and Pattern Recognition, pp. 248\u2013255. IEEE (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"12_CR9","unstructured":"de\u00a0Faria, A.C.A.M., et al.: Visual question answering: a survey on techniques and common trends in recent literature. arXiv preprint arXiv:2305.11033 (2023)"},{"key":"12_CR10","doi-asserted-by":"crossref","unstructured":"Gui, L., Wang, B., Huang, Q., Hauptmann, A., Bisk, Y., Gao, J.: Kat: a knowledge augmented transformer for vision-and-language. arXiv preprint arXiv:2112.08614 (2021)","DOI":"10.18653\/v1\/2022.naacl-main.70"},{"key":"12_CR11","doi-asserted-by":"crossref","unstructured":"Gupta, D., Lenka, P., Ekbal, A., Bhattacharyya, P.: A unified framework for multilingual and code-mixed visual question answering. In: Proceedings of the 1st Conference of the Asia-Pacific Chapter of the Association for Computational Linguistics and the 10th International Joint Conference on Natural Language Processing, pp. 900\u2013913 (2020)","DOI":"10.18653\/v1\/2020.aacl-main.90"},{"key":"12_CR12","first-page":"15908","volume":"34","author":"K Han","year":"2021","unstructured":"Han, K., Xiao, A., Wu, E., Guo, J., Xu, C., Wang, Y.: Transformer in transformer. Adv. Neural. Inf. Process. Syst. 34, 15908\u201315919 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"12_CR13","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"12_CR14","doi-asserted-by":"crossref","unstructured":"Islam, S.S., Auntor, R.A., Islam, M., Anik, M.Y.H., Islam, A.A.A., Noor, J.: Note: Towards devising an efficient VQA in the Bengali language. In: Proceedings of the 5th ACM SIGCAS\/SIGCHI Conference on Computing and Sustainable Societies, pp. 632\u2013637 (2022)","DOI":"10.1145\/3530190.3534837"},{"key":"12_CR15","doi-asserted-by":"crossref","unstructured":"Kafle, K., Kanan, C.: An analysis of visual question answering algorithms. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1965\u20131973 (2017)","DOI":"10.1109\/ICCV.2017.217"},{"key":"12_CR16","doi-asserted-by":"crossref","unstructured":"Khan, A.U., Kuehne, H., Gan, C., Lobo, N.D.V., Shah, M.: Weakly supervised grounding for VQA in vision-language transformers. In: European Conference on Computer Vision, pp. 652\u2013670. Springer (2022)","DOI":"10.1007\/978-3-031-19833-5_38"},{"key":"12_CR17","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","volume":"123","author":"R Krishna","year":"2017","unstructured":"Krishna, R., et al.: Visual genome: connecting language and vision using crowdsourced dense image annotations. Int. J. Comput. Vision 123, 32\u201373 (2017)","journal-title":"Int. J. Comput. Vision"},{"key":"12_CR18","doi-asserted-by":"crossref","unstructured":"Mishra, A., Anand, A., Guha, P.: Multi-stage attention based visual question answering. In: 2020 25th International Conference on Pattern Recognition (ICPR), pp. 9407\u20139414. IEEE (2021)","DOI":"10.1109\/ICPR48806.2021.9413173"},{"issue":"1","key":"12_CR19","doi-asserted-by":"publisher","first-page":"81","DOI":"10.1109\/TAI.2022.3160418","volume":"4","author":"A Mishra","year":"2022","unstructured":"Mishra, A., Anand, A., Guha, P.: Dual attention and question categorization-based visual question answering. IEEE Trans. Artif. Intell. 4(1), 81\u201391 (2022)","journal-title":"IEEE Trans. Artif. Intell."},{"key":"12_CR20","doi-asserted-by":"crossref","unstructured":"Parida, S., et al.: HAVQA: a dataset for visual question answering and multimodal research in Hausa language. arXiv preprint arXiv:2305.17690 (2023)","DOI":"10.18653\/v1\/2023.findings-acl.646"},{"key":"12_CR21","doi-asserted-by":"crossref","unstructured":"Pathak, D., Nandi, S., Sarmah, P.: ASPOS: assamese part of speech tagger using deep learning approach. In: 2022 IEEE\/ACS 19th International Conference on Computer Systems and Applications (AICCSA), pp.\u00a01\u20138. IEEE (2022)","DOI":"10.1109\/AICCSA56895.2022.10017934"},{"key":"12_CR22","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster R-CNN: towards real-time object detection with region proposal networks. In: Advances in Neural Information Processing Systems, vol. 28 (2015)"},{"key":"12_CR23","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2021.107408","volume":"230","author":"D Song","year":"2021","unstructured":"Song, D., Ma, S., Sun, Z., Yang, S., Liao, L.: KVL-Bert: Knowledge enhanced visual-and-linguistic Bert for visual commonsense reasoning. Knowl.-Based Syst. 230, 107408 (2021)","journal-title":"Knowl.-Based Syst."},{"key":"12_CR24","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"12_CR25","doi-asserted-by":"crossref","unstructured":"Zhu, Z., Yu, J., Wang, Y., Sun, Y., Hu, Y., Wu, Q.: MUCKO: multi-layer cross-modal knowledge reasoning for fact-based visual question answering. arXiv preprint arXiv:2006.09073 (2020)","DOI":"10.24963\/ijcai.2020\/153"}],"container-title":["Communications in Computer and Information Science","Computer Vision and Image Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-93697-5_12","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T23:29:46Z","timestamp":1782948586000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-93697-5_12"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,24]]},"ISBN":["9783031936968","9783031936975"],"references-count":25,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-93697-5_12","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"value":"1865-0929","type":"print"},{"value":"1865-0937","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,7,24]]},"assertion":[{"value":"24 July 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"CVIP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Computer Vision and Image Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Chennai","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"India","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 December 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 December 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"9","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"cvip2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/cvip2024.iiitdm.ac.in\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}