{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,7]],"date-time":"2026-05-07T15:16:50Z","timestamp":1778167010660,"version":"3.51.4"},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2023,2,16]],"date-time":"2023-02-16T00:00:00Z","timestamp":1676505600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,2,16]],"date-time":"2023-02-16T00:00:00Z","timestamp":1676505600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SN COMPUT. SCI."],"DOI":"10.1007\/s42979-023-01671-x","type":"journal-article","created":{"date-parts":[[2023,2,16]],"date-time":"2023-02-16T16:02:31Z","timestamp":1676563351000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":10,"title":["A Visual Attention-Based Model for Bengali Image Captioning"],"prefix":"10.1007","volume":"4","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8588-1913","authenticated-orcid":false,"given":"Bidyut","family":"Das","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ratnabali","family":"Pal","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mukta","family":"Majumder","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Santanu","family":"Phadikar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Arif Ahmed","family":"Sekh","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,2,16]]},"reference":[{"key":"1671_CR1","doi-asserted-by":"publisher","DOI":"10.1016\/j.compeleceng.2021.107114","volume":"92","author":"SK Mishra","year":"2021","unstructured":"Mishra SK, Dhir R, Saha S, Bhattacharyya P, Singh AK. Image captioning in hindi language using transformer networks. Computers & Electrical Engineering. 2021;92: 107114.","journal-title":"Computers & Electrical Engineering"},{"key":"1671_CR2","doi-asserted-by":"crossref","unstructured":"Das B, Sekh AA, Majumder M, Phadikar S, Abid: Attention-based bengali image description. In: Proceedings of the 3rd International Conference on Communication, Devices and Computing, 2022;305\u2013314 . Springer","DOI":"10.1007\/978-981-16-9154-6_29"},{"key":"1671_CR3","doi-asserted-by":"crossref","unstructured":"Vinyals O, Toshev A, Bengio S, Erhan, D, Show and tell: A neural image caption generator. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2015;3156\u20133164","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"1671_CR4","doi-asserted-by":"crossref","unstructured":"Johnson J, Karpathy A, Fei-Fei L, Densecap: Fully convolutional localization networks for dense captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016;4565\u20134574","DOI":"10.1109\/CVPR.2016.494"},{"key":"1671_CR5","doi-asserted-by":"crossref","unstructured":"Karpathy A, Fei-Fei L, Deep visual-semantic alignments for generating image descriptions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2015;3128\u20133137","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"1671_CR6","doi-asserted-by":"crossref","unstructured":"You Q, Jin H, Wang Z, Fang C, Luo J, Image captioning with semantic attention. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016;4651\u20134659","DOI":"10.1109\/CVPR.2016.503"},{"key":"1671_CR7","doi-asserted-by":"publisher","first-page":"853","DOI":"10.1613\/jair.3994","volume":"47","author":"M Hodosh","year":"2013","unstructured":"Hodosh M, Young P, Hockenmaier J. Framing image description as a ranking task: Data, models and evaluation metrics. Journal of Artificial Intelligence Research. 2013;47:853\u201399.","journal-title":"Journal of Artificial Intelligence Research"},{"key":"1671_CR8","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1162\/tacl_a_00166","volume":"2","author":"P Young","year":"2014","unstructured":"Young P, Lai A, Hodosh M, Hockenmaier J. From image descriptions to visual denotations: New similarity metrics for semantic inference over event descriptions. Transactions of the Association for Computational Linguistics. 2014;2:67\u201378.","journal-title":"Transactions of the Association for Computational Linguistics"},{"key":"1671_CR9","doi-asserted-by":"crossref","unstructured":"Lin T-Y, Maire M, Belongie S, Hays J, Perona P, Ramanan D, Doll\u00e1r P, Zitnick CL, Microsoft coco: Common objects in context. In: European Conference on Computer Vision, 2014;740\u2013755. Springer","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"1671_CR10","doi-asserted-by":"crossref","unstructured":"Gomez-Garay A, Raducanu B, Salas J, Dense captioning of natural scenes in spanish. In: Mexican Conference on Pattern Recognition, 2018;145\u2013154 . Springer","DOI":"10.1007\/978-3-319-92198-3_15"},{"key":"1671_CR11","doi-asserted-by":"publisher","first-page":"291","DOI":"10.1016\/j.neucom.2018.05.080","volume":"311","author":"S Bai","year":"2018","unstructured":"Bai S, An S. A survey on automatic image caption generation. Neurocomputing. 2018;311:291\u2013304.","journal-title":"Neurocomputing"},{"issue":"6","key":"1671_CR12","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3295748","volume":"51","author":"MZ Hossain","year":"2019","unstructured":"Hossain MZ, Sohel F, Shiratuddin MF, Laga H. A comprehensive survey of deep learning for image captioning. ACM Computing Surveys (CsUR). 2019;51(6):1\u201336.","journal-title":"ACM Computing Surveys (CsUR)"},{"key":"1671_CR13","unstructured":"Xu K, Ba J, Kiros R, Cho K, Courville A, Salakhudinov R, Zemel R, Bengio Y, Show, attend and tell: Neural image caption generation with visual attention. In: International Conference on Machine Learning, 2015;2048\u20132057 . PMLR"},{"key":"1671_CR14","doi-asserted-by":"crossref","unstructured":"Lu J, Xiong C, Parikh D, Socher R, Knowing when to look: Adaptive attention via a visual sentinel for image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2017;375\u2013383","DOI":"10.1109\/CVPR.2017.345"},{"key":"1671_CR15","doi-asserted-by":"crossref","unstructured":"Chen Q, Li W, Lei Y, Liu X, He Y, Learning to adapt credible knowledge in cross-lingual sentiment analysis. In: Proceedings of the 53rd Annual Meeting of the ACL and the 7th International Joint Conference on NLP (Volume 1: Long Papers), 2015;419\u2013429","DOI":"10.3115\/v1\/P15-1041"},{"key":"1671_CR16","doi-asserted-by":"crossref","unstructured":"Anderson P, He X, Buehler C, Teney D, Johnson M, Gould S, Zhang L, Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018;6077\u20136086","DOI":"10.1109\/CVPR.2018.00636"},{"key":"1671_CR17","doi-asserted-by":"crossref","unstructured":"Miyazaki T, Shimizu N, Cross-lingual image caption generation. In: Proceedings of the 54th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), 2016;1780\u20131790","DOI":"10.18653\/v1\/P16-1168"},{"key":"1671_CR18","doi-asserted-by":"crossref","unstructured":"Yoshikawa Y, Shigeto Y, Takeuchi A, Stair captions: Constructing a large-scale japanese image caption dataset. 2017, arXiv preprint arXiv:1705.00823","DOI":"10.18653\/v1\/P17-2066"},{"key":"1671_CR19","doi-asserted-by":"crossref","unstructured":"Rathi A, Deep learning apporach for image captioning in hindi language. In: 2020 International Conference on Computer, Electrical & Communication Engineering (ICCECE), 2020;1\u20138 . IEEE","DOI":"10.1109\/ICCECE48148.2020.9223087"},{"key":"1671_CR20","doi-asserted-by":"crossref","unstructured":"Li X, Lan W, Dong J, Liu H, Adding chinese captions to images. In: Proceedings of the 2016 ACM on International Conference on Multimedia Retrieval, 2016;271\u2013275","DOI":"10.1145\/2911996.2912049"},{"issue":"9","key":"1671_CR21","doi-asserted-by":"publisher","first-page":"2347","DOI":"10.1109\/TMM.2019.2896494","volume":"21","author":"X Li","year":"2019","unstructured":"Li X, Xu C, Wang X, Lan W, Jia Z, Yang G, Xu J. Coco-cn for cross-lingual image tagging, captioning, and retrieval. IEEE Trans Multimedia. 2019;21(9):2347\u201360.","journal-title":"IEEE Trans Multimedia"},{"key":"1671_CR22","doi-asserted-by":"crossref","unstructured":"Lan W, Li X, Dong J, Fluency-guided cross-lingual image captioning. In: Proceedings of the 25th ACM International Conference on Multimedia, 2017;1549\u20131557","DOI":"10.1145\/3123266.3123366"},{"key":"1671_CR23","doi-asserted-by":"crossref","unstructured":"Zeng X, Wang X, Add english to image chinese captioning. In: 2017 IEEE 2nd International Conference on Cloud Computing and Big Data Analysis (ICCCBDA), 2017;333\u2013338 . IEEE","DOI":"10.1109\/ICCCBDA.2017.7951934"},{"key":"1671_CR24","unstructured":"Elliott D, Frank S, Hasler E, Multilingual image description with neural sequence models. 2015, arXiv preprint arXiv:1510.04709"},{"key":"1671_CR25","doi-asserted-by":"crossref","unstructured":"Elliott D, Frank S, Sima\u2019an K, Specia L, Multi30k: Multilingual english-german image descriptions.2016, arXiv preprint arXiv:1605.00459","DOI":"10.18653\/v1\/W16-3210"},{"key":"1671_CR26","doi-asserted-by":"crossref","unstructured":"van Miltenburg E, Elliott D, Vossen P, Cross-linguistic differences and similarities in image descriptions. 2017, arXiv preprint arXiv:1707.01736","DOI":"10.18653\/v1\/W17-3503"},{"key":"1671_CR27","doi-asserted-by":"crossref","unstructured":"Kamal AH, Jishan MA, Mansoor N, Textmage: The automated bangla caption generator based on deep learning. In: 2020 International Conference on Decision Aid Sciences and Application (DASA), 2020;822\u2013826 . IEEE","DOI":"10.1109\/DASA51403.2020.9317108"},{"key":"1671_CR28","unstructured":"Khan MF, Shifath S, Islam M, et al. Improved bengali image captioning via deep convolutional neural network based encoder-decoder model. 2021, arXiv preprint arXiv:2102.07192"},{"issue":"2","key":"1671_CR29","doi-asserted-by":"publisher","first-page":"698","DOI":"10.14569\/IJACSA.2021.0120287","volume":"12","author":"M Humaira","year":"2021","unstructured":"Humaira M, Paul S, Jim M, Ami AS, Shah FM. A hybridized deep learning method for bengali image captioning. IJACSA. 2021;12(2):698\u2013707.","journal-title":"IJACSA"},{"key":"1671_CR30","unstructured":"Eddin Za\u2019ter M, Talaftha B, Bench-marking and improving arabic automatic image captioning through the use of multi-task learning paradigm. arXiv e-prints, 2202 (2022)"},{"key":"1671_CR31","unstructured":"Mansoor N, Kamal AH, Mohammed N, Momen S, Rahman MM, Banglalekhaimagecaptions, mendeley data 2019. (Date last accessed 15-July-2014). http:\/\/dx.doi.org\/10.17632\/rxxch9vw59.2"},{"issue":"2","key":"1671_CR32","doi-asserted-by":"publisher","first-page":"757","DOI":"10.11591\/ijeecs.v21.i2.pp757-767","volume":"21","author":"MA Jishan","year":"2021","unstructured":"Jishan MA, Mahmud KR, Al Azad AK, Ahmmad MR, Rashid BP, Alam MS. Bangla language textual image description by hybrid neural network model. Indonesian Journal of Electrical Engineering and Computer Science. 2021;21(2):757\u201367.","journal-title":"Indonesian Journal of Electrical Engineering and Computer Science"},{"key":"1671_CR33","doi-asserted-by":"crossref","unstructured":"Szegedy C, Vanhoucke V, Ioffe S, Shlens J, Wojna Z, Rethinking the inception architecture for computer vision. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016;2818\u20132826","DOI":"10.1109\/CVPR.2016.308"},{"issue":"3","key":"1671_CR34","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"Russakovsky O, Deng J, Su H, Krause J, Satheesh S, Ma S, Huang Z, Karpathy A, Khosla A, Bernstein M, et al. Imagenet large scale visual recognition challenge. Int J Comput Vision. 2015;115(3):211\u201352.","journal-title":"Int J Comput Vision"},{"key":"1671_CR35","doi-asserted-by":"crossref","unstructured":"Cho K, van Merri\u00ebnboer B, Gulcehre C, Bahdanau D, Bougares F, Schwenk H, Bengio Y, Learning phrase representations using rnn encoder\u2013decoder for statistical machine translation. In: Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP), 2014;1724\u20131734","DOI":"10.3115\/v1\/D14-1179"},{"key":"1671_CR36","doi-asserted-by":"crossref","unstructured":"Papineni K., Roukos S, Ward T, Zhu W-J, Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, 2002;311\u2013318","DOI":"10.3115\/1073083.1073135"},{"key":"1671_CR37","unstructured":"Simonyan K, Zisserman A, Very deep convolutional networks for large-scale image recognition. 2014, arXiv preprint arXiv:1409.1556"},{"key":"1671_CR38","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J, Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016;770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"1671_CR39","doi-asserted-by":"crossref","unstructured":"Tanti M, Gatt A, Camilleri K, What is the role of recurrent neural networks (rnns) in an image caption generator? In: Proceedings of the 10th International Conference on Natural Language Generation, 2017;51\u201360","DOI":"10.18653\/v1\/W17-3506"},{"key":"1671_CR40","doi-asserted-by":"publisher","first-page":"636","DOI":"10.1016\/j.procs.2019.06.100","volume":"154","author":"M Rahman","year":"2019","unstructured":"Rahman M, Mohammed N, Mansoor N, Momen S. Chittron: An automatic bangla image captioning system. Procedia Computer Science. 2019;154:636\u201342.","journal-title":"Procedia Computer Science"}],"container-title":["SN Computer Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42979-023-01671-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s42979-023-01671-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42979-023-01671-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,14]],"date-time":"2024-10-14T11:54:52Z","timestamp":1728906892000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s42979-023-01671-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,2,16]]},"references-count":40,"journal-issue":{"issue":"2","published-online":{"date-parts":[[2023,3]]}},"alternative-id":["1671"],"URL":"https:\/\/doi.org\/10.1007\/s42979-023-01671-x","relation":{},"ISSN":["2661-8907"],"issn-type":[{"value":"2661-8907","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,2,16]]},"assertion":[{"value":"27 June 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 January 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 February 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"All authors read and approved the final manuscript.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}}],"article-number":"208"}}