{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,22]],"date-time":"2026-03-22T22:54:09Z","timestamp":1774220049740,"version":"3.50.1"},"reference-count":25,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SIViP"],"published-print":{"date-parts":[[2026,3]]},"DOI":"10.1007\/s11760-026-05233-5","type":"journal-article","created":{"date-parts":[[2026,3,9]],"date-time":"2026-03-09T18:57:39Z","timestamp":1773082659000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Towards explainable AI: multi-modal transformer for video-based image description generation"],"prefix":"10.1007","volume":"20","author":[{"given":"Lakshita","family":"Agarwal","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bindu","family":"Verma","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,3,9]]},"reference":[{"issue":"6","key":"5233_CR1","first-page":"1","volume":"52","author":"N Aafaq","year":"2019","unstructured":"Aafaq, N., Mian, A., Liu, W., Gilani, S.Z., Shah, M.: Video description: a survey of methods, datasets, and evaluation metrics. ACM Comput. Sur. (CSUR) 52(6), 1\u201337 (2019)","journal-title":"ACM Comput. Sur. (CSUR)"},{"key":"5233_CR2","doi-asserted-by":"crossref","unstructured":"Li, T., Wang, H., Li, X., Liao, W., He, T., Peng, P.: Generative planning with 3d-vision language pre-training for end-to-end autonomous driving. arXiv:2501.08861 (2025)","DOI":"10.1609\/aaai.v39i5.32524"},{"issue":"9","key":"5233_CR3","doi-asserted-by":"publisher","first-page":"28077","DOI":"10.1007\/s11042-023-16560-x","volume":"83","author":"L Agarwal","year":"2024","unstructured":"Agarwal, L., Verma, B.: From methods to datasets: a survey on image-caption generators. Multimed. Tools Appl. 83(9), 28077\u201328123 (2024)","journal-title":"Multimed. Tools Appl."},{"key":"5233_CR4","doi-asserted-by":"crossref","unstructured":"Shoman, M., Wang, D., Aboah, A., Abdel-Aty, M.: Enhancing traffic safety with parallel dense video captioning for end-to-end event analysis, in Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7125\u20137133 (2024)","DOI":"10.1109\/CVPRW63382.2024.00707"},{"issue":"7","key":"5233_CR5","doi-asserted-by":"publisher","first-page":"7805","DOI":"10.1109\/TITS.2021.3072970","volume":"23","author":"Y Li","year":"2021","unstructured":"Li, Y., Wu, C., Li, L., Liu, Y., Zhu, J.: Caption generation from road images for traffic scene modeling. IEEE Trans. Intell. Transp. Syst. 23(7), 7805\u20137816 (2021)","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"issue":"6","key":"5233_CR6","doi-asserted-by":"publisher","first-page":"8268","DOI":"10.1007\/s11227-021-04151-2","volume":"78","author":"C Gou","year":"2022","unstructured":"Gou, C., Zhou, Y., Li, D.: Driver attention prediction based on convolution and transformers. J. Supercomput. 78(6), 8268\u20138284 (2022)","journal-title":"J. Supercomput."},{"key":"5233_CR7","doi-asserted-by":"crossref","unstructured":"Lei, J., Wang, L., Shen, Y., Yu, D., Berg, T.L., Bansal, M.: Mart: Memory-augmented recurrent transformer for coherent video paragraph captioning. arXiv:2005.05402 (2020)","DOI":"10.18653\/v1\/2020.acl-main.233"},{"key":"5233_CR8","doi-asserted-by":"crossref","unstructured":"Sun, C., Myers, A., Vondrick, C., Murphy, K., Schmid, C.: Videobert: A joint model for video and language representation learning, in Proceedings of the IEEE\/CVF international conference on computer vision, pp. 7464\u20137473 (2019)","DOI":"10.1109\/ICCV.2019.00756"},{"key":"5233_CR9","doi-asserted-by":"crossref","unstructured":"Lei, J., Li, L., Zhou, L., Gan, Z., Berg, T.L., Bansal, M., Liu, J.: Less is more: Clipbert for video-and-language learning via sparse sampling, in Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 7331\u20137341 (2021)","DOI":"10.1109\/CVPR46437.2021.00725"},{"key":"5233_CR10","unstructured":"Chen, D., Dolan, W.B.: Collecting highly parallel data for paraphrase evaluation, in Proceedings of the 49th annual meeting of the association for computational linguistics: human language technologies, pp. 190\u2013200 (2011)"},{"key":"5233_CR11","doi-asserted-by":"crossref","unstructured":"Xu, J., Mei, T., Yao, T., Rui, Y.: Msr-vtt: A large video description dataset for bridging video and language, in Proceedings of the IEEE conference on computer vision and pattern recognition , pp. 5288\u20135296 (2016)","DOI":"10.1109\/CVPR.2016.571"},{"key":"5233_CR12","doi-asserted-by":"crossref","unstructured":"Kim, J., Rohrbach, A., Darrell, T., Canny, J., Akata, Z.: Textual explanations for self-driving vehicles, in Proceedings of the European conference on computer vision (ECCV), pp. 563\u2013578 (2018)","DOI":"10.1007\/978-3-030-01216-8_35"},{"key":"5233_CR13","doi-asserted-by":"crossref","unstructured":"Cui, C., Ma, Y., Cao, X., Ye, W., Zhou, Y., Liang, K., Chen, J., Lu, J., Yang, Z., Liao, K.D., et al.: A survey on multimodal large language models for autonomous driving, in Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 958\u2013979 (2024)","DOI":"10.1109\/WACVW60836.2024.00106"},{"key":"5233_CR14","doi-asserted-by":"crossref","unstructured":"Yang, A., Nagrani, A., Seo, P.H., Miech, A., Pont-Tuset, J., Laptev, I., Sivic, J., Schmid, C.: Vid2seq: Large-scale pretraining of a visual language model for dense video captioning, in Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 10714\u201310726 (2023)","DOI":"10.1109\/CVPR52729.2023.01032"},{"key":"5233_CR15","doi-asserted-by":"crossref","unstructured":"Maaz, M., Rasheed, H., Khan, S., Khan, F.S.: Video-chatgpt: Towards detailed video understanding via large vision and language models. arXiv:2306.05424 (2023)","DOI":"10.18653\/v1\/2024.acl-long.679"},{"key":"5233_CR16","unstructured":"Cheng, Z., Leng, S., Zhang, H., Xin, Y., Li, X., Chen, G., Zhu, Y., Zhang, W., Luo, Z., Zhao, D., et al.: Videollama 2: Advancing spatial-temporal modeling and audio understanding in video-llms. arXiv: 2406.07476 (2024)"},{"key":"5233_CR17","doi-asserted-by":"crossref","unstructured":"Huang, B., Wang, X., Chen, H., Song, Z., Zhu, W.: Vtimellm: Empower llm to grasp video moments, in Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14271\u201314280 (2024)","DOI":"10.1109\/CVPR52733.2024.01353"},{"key":"5233_CR18","doi-asserted-by":"crossref","unstructured":"Pan, Y., Yao, T., Li, Y., Mei, T.: X-linear attention networks for image captioning, in Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 10971\u201310980 (2020)","DOI":"10.1109\/CVPR42600.2020.01098"},{"issue":"6","key":"5233_CR19","doi-asserted-by":"publisher","first-page":"55","DOI":"10.55524\/ijirem.2023.10.6.8","volume":"10","author":"I Vasireddy","year":"2023","unstructured":"Vasireddy, I., HimaBindu, G., Ratnamala, B.: Transformative fusion: vision transformers and gpt-2 unleashing new frontiers in image captioning within image processing. Intl. J. Innovative Res. Eng. Mgmnt 10(6), 55\u201359 (2023)","journal-title":"Intl. J. Innovative Res. Eng. Mgmnt"},{"key":"5233_CR20","doi-asserted-by":"crossref","unstructured":"Li, X., Yin, X., Li, C., Zhang, P., Hu, X., Zhang, L., Wang, L., Hu, H., Dong, L., Wei, F., et al.: Oscar: Object-semantics aligned pre-training for vision-language tasks, in Computer Vision-ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, Proceedings, Part XXX 16 (Springer, 2020), pp. 121\u2013137 (2020)","DOI":"10.1007\/978-3-030-58577-8_8"},{"key":"5233_CR21","doi-asserted-by":"crossref","unstructured":"Huang, L., Wang, W., Chen, J., Wei, X.Y.: Attention on attention for image captioning, in Proceedings of the IEEE\/CVF international conference on computer vision, pp. 4634\u20134643 (2019)","DOI":"10.1109\/ICCV.2019.00473"},{"key":"5233_CR22","doi-asserted-by":"crossref","unstructured":"Vinyals, O., Toshev, A., Bengio, S., Erhan, D.: Show and tell: A neural image caption generator, in Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 3156\u20133164 (2015)","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"5233_CR23","doi-asserted-by":"crossref","unstructured":"Cornia, M., Stefanini, M., Baraldi, L., Cucchiara, R.: Meshed-memory transformer for image captioning, in Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 10578\u201310587 (2020)","DOI":"10.1109\/CVPR42600.2020.01059"},{"key":"5233_CR24","doi-asserted-by":"crossref","unstructured":"Krishna, R., Hata, K., Ren, F., Fei-Fei, L., Carlos Niebles, J.: Dense-captioning events in videos, in Proceedings of the IEEE international conference on computer vision, pp. 706\u2013715 (2017)","DOI":"10.1109\/ICCV.2017.83"},{"key":"5233_CR25","doi-asserted-by":"crossref","unstructured":"Aafaq, N., Akhtar, N., Liu, W., Gilani, S.Z., Mian, A.: Spatio-temporal dynamics and semantic attribute enriched visual encoding for video captioning, in Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 12487\u201312496 (2019)","DOI":"10.1109\/CVPR.2019.01277"}],"container-title":["Signal, Image and Video Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-026-05233-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11760-026-05233-5","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-026-05233-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,22]],"date-time":"2026-03-22T22:15:22Z","timestamp":1774217722000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11760-026-05233-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3]]},"references-count":25,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2026,3]]}},"alternative-id":["5233"],"URL":"https:\/\/doi.org\/10.1007\/s11760-026-05233-5","relation":{},"ISSN":["1863-1703","1863-1711"],"issn-type":[{"value":"1863-1703","type":"print"},{"value":"1863-1711","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3]]},"assertion":[{"value":"8 April 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 October 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 February 2026","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 March 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing Interests"}}],"article-number":"141"}}