{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T17:36:07Z","timestamp":1776965767897,"version":"3.51.4"},"reference-count":69,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T00:00:00Z","timestamp":1769904000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T00:00:00Z","timestamp":1769904000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"the National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["Nos. 62371144"],"award-info":[{"award-number":["Nos. 62371144"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2026,2]]},"DOI":"10.1007\/s00371-026-04353-w","type":"journal-article","created":{"date-parts":[[2026,2,2]],"date-time":"2026-02-02T09:12:30Z","timestamp":1770023550000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Enhancing video captioning with contextual anchor-guided semantic modeling"],"prefix":"10.1007","volume":"42","author":[{"given":"Zhonghua","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lina","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xichun","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Thomas","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yifeng","family":"Tan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yijun","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,2,2]]},"reference":[{"key":"4353_CR1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2020.107248","volume":"102","author":"W Wang","year":"2020","unstructured":"Wang, W., Huang, Y., Wang, L.: Long video question answering: a matching-guided attention model. Pattern Recogn. 102, 107248 (2020)","journal-title":"Pattern Recogn."},{"key":"4353_CR2","doi-asserted-by":"crossref","unstructured":"Song, J., Guo, Z., Gao, L., Liu, W., Zhang, D., Shen, H.T.: Hierarchical lstm with adjusted temporal attention for video captioning. arXiv preprint arXiv:1706.01231 (2017)","DOI":"10.24963\/ijcai.2017\/381"},{"key":"4353_CR3","doi-asserted-by":"publisher","first-page":"6821","DOI":"10.1109\/TMM.2022.3214776","volume":"25","author":"J Zhu","year":"2022","unstructured":"Zhu, J., Zhang, Q., Fei, L., Cai, R., Xie, Y., Sheng, B., Yang, X.: Fffn: Frame-by-frame feedback fusion network for video super-resolution. IEEE Trans. Multimedia 25, 6821\u20136835 (2022)","journal-title":"IEEE Trans. Multimedia"},{"key":"4353_CR4","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2022.108807","volume":"130","author":"T-H Chiang","year":"2022","unstructured":"Chiang, T.-H., Tseng, Y.-C., Tseng, Y.-C.: A multi-embedding neural model for incident video retrieval. Pattern Recogn. 130, 108807 (2022)","journal-title":"Pattern Recogn."},{"issue":"2","key":"4353_CR5","doi-asserted-by":"publisher","first-page":"880","DOI":"10.1109\/TCSVT.2021.3063423","volume":"32","author":"J Deng","year":"2021","unstructured":"Deng, J., Li, L., Zhang, B., Wang, S., Zha, Z., Huang, Q.: Syntax-guided hierarchical attention network for video captioning. IEEE Trans. Circuits Syst. Video Technol. 32(2), 880\u2013892 (2021)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"issue":"1","key":"4353_CR6","doi-asserted-by":"publisher","first-page":"17","DOI":"10.1109\/TCSVT.2020.3045735","volume":"32","author":"L Li","year":"2020","unstructured":"Li, L., Zhang, Y., Tang, S., Xie, L., Li, X., Tian, Q.: Adaptive spatial location with balanced loss for video captioning. IEEE Trans. Circuits Syst. Video Technol. 32(1), 17\u201330 (2020)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"issue":"9","key":"4353_CR7","doi-asserted-by":"publisher","first-page":"6293","DOI":"10.1109\/TCSVT.2022.3165934","volume":"32","author":"W Xu","year":"2022","unstructured":"Xu, W., Miao, Z., Yu, J., Tian, Y., Wan, L., Ji, Q.: Bridging video and text: A two-step polishing transformer for video captioning. IEEE Trans. Circuits Syst. Video Technol. 32(9), 6293\u20136307 (2022)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"4353_CR8","doi-asserted-by":"crossref","unstructured":"Liu, S., Li, A., Zhao, Y., Wang, J., Wang, Y.: Evcap: Element-aware video captioning. IEEE Trans. Circuits Syst. Video Technol. (2024)","DOI":"10.1109\/TCSVT.2024.3399933"},{"issue":"9","key":"4353_CR9","doi-asserted-by":"publisher","first-page":"4484","DOI":"10.1109\/TCSVT.2023.3277827","volume":"33","author":"B Wu","year":"2023","unstructured":"Wu, B., Liu, B., Huang, P., Bao, J., Xi, P., Yu, J.: Concept parser with multimodal graph learning for video captioning. IEEE Trans. Circuits Syst. Video Technol. 33(9), 4484\u20134495 (2023)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"4353_CR10","doi-asserted-by":"crossref","unstructured":"Jiang, W., Liu, L., Fang, Y., Cheng, Y., Peng, Y., Liu, Y.: Learning comprehensive visual grounding for video captioning. IEEE Trans. Circuits Syst. Video Technol. (2024)","DOI":"10.1109\/TCSVT.2024.3502621"},{"key":"4353_CR11","doi-asserted-by":"crossref","unstructured":"Zheng, Q., Wang, C., Tao, D.: Syntax-aware action targeting for video captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13096\u201313105 (2020)","DOI":"10.1109\/CVPR42600.2020.01311"},{"key":"4353_CR12","doi-asserted-by":"crossref","unstructured":"Ye, H., Li, G., Qi, Y., Wang, S., Huang, Q., Yang, M.-H.: Hierarchical modular network for video captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17939\u201317948 (2022)","DOI":"10.1109\/CVPR52688.2022.01741"},{"issue":"2","key":"4353_CR13","doi-asserted-by":"publisher","first-page":"1049","DOI":"10.1109\/TPAMI.2023.3327677","volume":"46","author":"G Li","year":"2024","unstructured":"Li, G., Ye, H., Qi, Y., Wang, S., Qing, L., Huang, Q., Yang, M.-H.: Learning hierarchical modular networks for video captioning. IEEE Trans. Pattern Anal. Mach. Intell. 46(2), 1049\u20131064 (2024)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"4353_CR14","doi-asserted-by":"publisher","first-page":"171","DOI":"10.1023\/A:1020346032608","volume":"50","author":"A Kojima","year":"2002","unstructured":"Kojima, A., Tamura, T., Fukunaga, K.: Natural language description of human activities from video images based on concept hierarchy of actions. Int. J. Comput. Vision 50, 171\u2013184 (2002)","journal-title":"Int. J. Comput. Vision"},{"key":"4353_CR15","doi-asserted-by":"crossref","unstructured":"Krishnamoorthy, N., Malkarnenkar, G., Mooney, R., Saenko, K., Guadarrama, S.: Generating natural-language video descriptions using text-mined knowledge. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 27, pp. 541\u2013547 (2013)","DOI":"10.1609\/aaai.v27i1.8679"},{"key":"4353_CR16","doi-asserted-by":"crossref","unstructured":"Pan, P., Xu, Z., Yang, Y., Wu, F., Zhuang, Y.: Hierarchical recurrent neural encoder for video representation with application to captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1029\u20131038 (2016)","DOI":"10.1109\/CVPR.2016.117"},{"key":"4353_CR17","doi-asserted-by":"crossref","unstructured":"Yao, L., Torabi, A., Cho, K., Ballas, N., Pal, C., Larochelle, H., Courville, A.: Describing videos by exploiting temporal structure. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 4507\u20134515 (2015)","DOI":"10.1109\/ICCV.2015.512"},{"issue":"12","key":"4353_CR18","doi-asserted-by":"publisher","first-page":"4312","DOI":"10.3390\/app10124312","volume":"10","author":"J Xu","year":"2020","unstructured":"Xu, J., Wei, H., Li, L., Fu, Q., Guo, J.: Video description model based on temporal-spatial and channel multi-attention mechanisms. Appl. Sci. 10(12), 4312 (2020)","journal-title":"Appl. Sci."},{"key":"4353_CR19","doi-asserted-by":"crossref","unstructured":"Wang, J., Wang, W., Huang, Y., Wang, L., Tan, T.: M3: Multimodal memory modelling for video captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7512\u20137520 (2018)","DOI":"10.1109\/CVPR.2018.00784"},{"key":"4353_CR20","doi-asserted-by":"crossref","unstructured":"Pei, W., Zhang, J., Wang, X., Ke, L., Shen, X., Tai, Y.-W.: Memory-attended recurrent network for video captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8347\u20138356 (2019)","DOI":"10.1109\/CVPR.2019.00854"},{"key":"4353_CR21","doi-asserted-by":"crossref","unstructured":"Aafaq, N., Akhtar, N., Liu, W., Gilani, S.Z., Mian, A.: Spatio-temporal dynamics and semantic attribute enriched visual encoding for video captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12487\u201312496 (2019)","DOI":"10.1109\/CVPR.2019.01277"},{"key":"4353_CR22","first-page":"170","volume":"459","author":"P Li","year":"2021","unstructured":"Li, P., Zhang, P., Xu, X.: Graph convolutional network meta-learning with multi-granularity pos guidance for video captioning. Neurocomputing 459, 170\u2013181 (2021)","journal-title":"Neurocomputing"},{"key":"4353_CR23","doi-asserted-by":"crossref","unstructured":"Zheng, Z., Wu, H., Wang, J., Lv, L., Bardou, D., Yu, G.: Vlca: vision-language feature enhancement with cross-attention learning for facial expression recognition. Expert Syst. Appl. 130292 (2025)","DOI":"10.1016\/j.eswa.2025.130292"},{"key":"4353_CR24","doi-asserted-by":"crossref","unstructured":"Zheng, Z., Wu, H., Lv, L., Zhang, C., Guo, H., Niu, S., Yu, G.: Multi-branch semantic alignment for few-shot image classification. Inform. Sci. 122676 (2025)","DOI":"10.1016\/j.ins.2025.122676"},{"key":"4353_CR25","doi-asserted-by":"crossref","unstructured":"Jin, T., Huang, S., Chen, M., Li, Y., Zhang, Z.: Sbat: Video captioning with sparse boundary-aware transformer. arXiv preprint arXiv:2007.11888 (2020)","DOI":"10.24963\/ijcai.2020\/88"},{"key":"4353_CR26","doi-asserted-by":"crossref","unstructured":"Peng, Y., Wang, C., Pei, Y., Li, Y.: Video captioning with global and local text attention. The Visual Computer, 1\u201312 (2022)","DOI":"10.1007\/s00371-021-02294-0"},{"key":"4353_CR27","unstructured":"Lian, L., Ding, Y., Ge, Y., Liu, S., Mao, H., Li, B., Pavone, M., Liu, M.-Y., Darrell, T., Yala, A., et al.: Describe anything: detailed localized image and video captioning. arXiv preprint arXiv:2504.16072 (2025)"},{"key":"4353_CR28","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2024.111138","volume":"159","author":"F Yuan","year":"2025","unstructured":"Yuan, F., Gu, S., Zhang, X., Fang, Z.: Fully exploring object relation interaction and hidden state attention for video captioning. Pattern Recogn. 159, 111138 (2025)","journal-title":"Pattern Recogn."},{"key":"4353_CR29","doi-asserted-by":"publisher","first-page":"621","DOI":"10.1007\/s11280-018-0531-z","volume":"22","author":"X Li","year":"2019","unstructured":"Li, X., Zhou, Z., Chen, L., Gao, L.: Residual attention-based lstm for video captioning. World Wide Web 22, 621\u2013636 (2019)","journal-title":"World Wide Web"},{"key":"4353_CR30","doi-asserted-by":"publisher","DOI":"10.1016\/j.asoc.2021.108332","volume":"117","author":"W Ji","year":"2022","unstructured":"Ji, W., Wang, R., Tian, Y., Wang, X.: An attention based dual learning approach for video captioning. Appl. Soft Comput. 117, 108332 (2022)","journal-title":"Appl. Soft Comput."},{"key":"4353_CR31","doi-asserted-by":"crossref","unstructured":"Sun, H., Li, S., Xi, Z., Wu, L.: Unified hierarchical contrastive learning for video captioning. Inform. Fusion 103856 (2025)","DOI":"10.1016\/j.inffus.2025.103856"},{"key":"4353_CR32","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2024.110898","volume":"157","author":"Y Su","year":"2025","unstructured":"Su, Y., Tan, Y., An, S., Xing, M., Feng, Z.: Semantic-driven dual consistency learning for weakly supervised video anomaly detection. Pattern Recogn. 157, 110898 (2025)","journal-title":"Pattern Recogn."},{"key":"4353_CR33","doi-asserted-by":"crossref","unstructured":"Zhou, L., Zhou, Y., Corso, J.J., Socher, R., Xiong, C.: End-to-end dense video captioning with masked transformer. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 8739\u20138748 (2018)","DOI":"10.1109\/CVPR.2018.00911"},{"key":"4353_CR34","doi-asserted-by":"crossref","unstructured":"Sun, C., Myers, A., Vondrick, C., Murphy, K., Schmid, C.: Videobert: A joint model for video and language representation learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7464\u20137473 (2019)","DOI":"10.1109\/ICCV.2019.00756"},{"key":"4353_CR35","doi-asserted-by":"crossref","unstructured":"Xie, S., Sun, C., Huang, J., Tu, Z., Murphy, K.: Rethinking spatiotemporal feature learning: Speed-accuracy trade-offs in video classification. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 305\u2013321 (2018)","DOI":"10.1007\/978-3-030-01267-0_19"},{"key":"4353_CR36","unstructured":"Child, R., Gray, S., Radford, A., Sutskever, I.: Generating long sequences with sparse transformers. arXiv preprint arXiv:1904.10509 (2019)"},{"key":"4353_CR37","doi-asserted-by":"crossref","unstructured":"Dai, Z., Yang, Z., Yang, Y., Carbonell, J., Le, Q.V., Salakhutdinov, R.: Transformer-xl: Attentive language models beyond a fixed-length context. arXiv preprint arXiv:1901.02860 (2019)","DOI":"10.18653\/v1\/P19-1285"},{"issue":"10","key":"4353_CR38","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3679203","volume":"20","author":"S Chen","year":"2024","unstructured":"Chen, S., Zhong, X., Zhang, Y., Zhu, L., Li, P., Yang, X., Sheng, B.: Action-aware linguistic skeleton optimization network for non-autoregressive video captioning. ACM Trans. Multimed. Comput. Commun. Appl. 20(10), 1\u201324 (2024)","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."},{"issue":"1","key":"4353_CR39","doi-asserted-by":"publisher","first-page":"591","DOI":"10.1007\/s00371-024-03350-1","volume":"41","author":"L Zheng","year":"2025","unstructured":"Zheng, L., Xu, W., Miao, Z., Qiu, X., Gong, S.: Restht: relation-enhanced spatial-temporal hierarchical transformer for video captioning. Vis. Comput. 41(1), 591\u2013604 (2025)","journal-title":"Vis. Comput."},{"issue":"2","key":"4353_CR40","doi-asserted-by":"publisher","first-page":"376","DOI":"10.1007\/s11227-024-06886-0","volume":"81","author":"H Wu","year":"2025","unstructured":"Wu, H., Zheng, Z., Lv, L., Zhang, C., Bardou, D., Niu, S., Yu, G.: Dara: distribution-aware representation alignment for semi-supervised domain adaptation in image classification. J. Supercomput. 81(2), 376 (2025)","journal-title":"J. Supercomput."},{"key":"4353_CR41","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2023.110412","volume":"266","author":"Z Zheng","year":"2023","unstructured":"Zheng, Z., Wu, H., Lv, L., Ye, H., Zhang, C., Yu, G.: Iccl: independent and correlative correspondence learning for few-shot image classification. Knowl.-Based Syst. 266, 110412 (2023)","journal-title":"Knowl.-Based Syst."},{"key":"4353_CR42","doi-asserted-by":"crossref","unstructured":"Gao, D., Zhou, L., Ji, L., Zhu, L., Yang, Y., Shou, M.Z.: Mist: Multi-modal iterative spatial-temporal transformer for long-form video question answering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14773\u201314783 (2023)","DOI":"10.1109\/CVPR52729.2023.01419"},{"key":"4353_CR43","doi-asserted-by":"crossref","unstructured":"Fang, Z., Gokhale, T., Banerjee, P., Baral, C., Yang, Y.: Video2commonsense: Generating commonsense descriptions to enrich video captioning. arXiv preprint arXiv:2003.05162 (2020)","DOI":"10.18653\/v1\/2020.emnlp-main.61"},{"key":"4353_CR44","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster r-cnn: Towards real-time object detection with region proposal networks. Advances in neural information processing systems 28 (2015)"},{"key":"4353_CR45","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: European Conference on Computer Vision, pp. 213\u2013229 (2020). Springer","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"4353_CR46","doi-asserted-by":"crossref","unstructured":"Reimers, N., Gurevych, I.: Sentence-bert: Sentence embeddings using siamese bert-networks. arXiv preprint arXiv:1908.10084 (2019)","DOI":"10.18653\/v1\/D19-1410"},{"issue":"1","key":"4353_CR47","doi-asserted-by":"publisher","first-page":"229","DOI":"10.1109\/TMM.2019.2924576","volume":"22","author":"C Yan","year":"2019","unstructured":"Yan, C., Tu, Y., Wang, X., Zhang, Y., Hao, X., Zhang, Y., Dai, Q.: Stat: Spatial-temporal attention mechanism for video captioning. IEEE Trans. Multimedia 22(1), 229\u2013241 (2019)","journal-title":"IEEE Trans. Multimedia"},{"key":"4353_CR48","doi-asserted-by":"crossref","unstructured":"Cai, X., Lai, Q., Wang, Y., Wang, W., Sun, Z., Yao, Y.: Poly kernel inception network for remote sensing detection. arxiv 2024. arXiv preprint arXiv:2403.06258","DOI":"10.1109\/CVPR52733.2024.02617"},{"key":"4353_CR49","doi-asserted-by":"crossref","unstructured":"Zhang, R., Zhu, F., Liu, J., Liu, G.: Depth-wise separable convolutions and multi-level pooling for an efficient spatial cnn-based steganalysis. IEEE Trans. Inf. Forensics Secur. 15, 1138\u20131150 (2019)","DOI":"10.1109\/TIFS.2019.2936913"},{"key":"4353_CR50","unstructured":"Chen, D., Dolan, W.B.: Collecting highly parallel data for paraphrase evaluation. In: Proceedings of the 49th Annual Meeting of the Association for Computational Linguistics: Human Language Technologies, pp. 190\u2013200 (2011)"},{"key":"4353_CR51","doi-asserted-by":"crossref","unstructured":"Xu, J., Mei, T., Yao, T., Rui, Y.: Msr-vtt: A large video description dataset for bridging video and language. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5288\u20135296 (2016)","DOI":"10.1109\/CVPR.2016.571"},{"key":"4353_CR52","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.-J.: Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, pp. 311\u2013318 (2002)","DOI":"10.3115\/1073083.1073135"},{"key":"4353_CR53","unstructured":"Banerjee, S., Lavie, A.: Meteor: An automatic metric for mt evaluation with improved correlation with human judgments. In: Proceedings of the Acl Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation And\/or Summarization, pp. 65\u201372 (2005)"},{"key":"4353_CR54","unstructured":"Lin, C.-Y.: Rouge: A package for automatic evaluation of summaries. In: Text Summarization Branches Out, pp. 74\u201381 (2004)"},{"key":"4353_CR55","doi-asserted-by":"crossref","unstructured":"Vedantam, R., Lawrence\u00a0Zitnick, C., Parikh, D.: Cider: Consensus-based image description evaluation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4566\u20134575 (2015)","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"4353_CR56","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","volume":"123","author":"R Krishna","year":"2017","unstructured":"Krishna, R., Zhu, Y., Groth, O., Johnson, J., Hata, K., Kravitz, J., Chen, S., Kalantidis, Y., Li, L.-J., Shamma, D.A., et al.: Visual genome: Connecting language and vision using crowdsourced dense image annotations. Int. J. Comput. Vision 123, 32\u201373 (2017)","journal-title":"Int. J. Comput. Vision"},{"key":"4353_CR57","doi-asserted-by":"crossref","unstructured":"Chen, S., Jiang, Y.-G.: Motion guided region message passing for video captioning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1543\u20131552 (2021)","DOI":"10.1109\/ICCV48922.2021.00157"},{"key":"4353_CR58","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Shi, Y., Yuan, C., Li, B., Wang, P., Hu, W., Zha, Z.-J.: Object relational graph with teacher-recommended learning for video captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13278\u201313288 (2020)","DOI":"10.1109\/CVPR42600.2020.01329"},{"key":"4353_CR59","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Ioffe, S., Vanhoucke, V., Alemi, A.: Inception-v4, inception-resnet and the impact of residual connections on learning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 31 (2017)","DOI":"10.1609\/aaai.v31i1.11231"},{"key":"4353_CR60","doi-asserted-by":"crossref","unstructured":"Hara, K., Kataoka, H., Satoh, Y.: Can spatiotemporal 3d cnns retrace the history of 2d cnns and imagenet? In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6546\u20136555 (2018)","DOI":"10.1109\/CVPR.2018.00685"},{"key":"4353_CR61","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2020.107702","volume":"111","author":"Y Tu","year":"2021","unstructured":"Tu, Y., Zhou, C., Guo, J., Gao, S., Yu, Z.: Enhancing the alignment between target words and corresponding frames for video captioning. Pattern Recogn. 111, 107702 (2021)","journal-title":"Pattern Recogn."},{"key":"4353_CR62","doi-asserted-by":"crossref","unstructured":"Ryu, H., Kang, S., Kang, H., Yoo, C.D.: Semantic grouping network for video captioning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, pp. 2514\u20132522 (2021)","DOI":"10.1609\/aaai.v35i3.16353"},{"key":"4353_CR63","doi-asserted-by":"publisher","first-page":"202","DOI":"10.1109\/TIP.2021.3120867","volume":"31","author":"L Gao","year":"2021","unstructured":"Gao, L., Lei, Y., Zeng, P., Song, J., Wang, M., Shen, H.T.: Hierarchical representation network with auxiliary tasks for video captioning and video question answering. IEEE Trans. Image Process. 31, 202\u2013215 (2021)","journal-title":"IEEE Trans. Image Process."},{"key":"4353_CR64","doi-asserted-by":"publisher","first-page":"2726","DOI":"10.1109\/TIP.2022.3158546","volume":"31","author":"L Li","year":"2022","unstructured":"Li, L., Gao, X., Deng, J., Tu, Y., Zha, Z.-J., Huang, Q.: Long short-term relation transformer with global gating for video captioning. IEEE Trans. Image Process. 31, 2726\u20132738 (2022)","journal-title":"IEEE Trans. Image Process."},{"key":"4353_CR65","doi-asserted-by":"crossref","unstructured":"Zhong, X., Li, Z., Chen, S., Jiang, K., Chen, C., Ye, M.: Refined semantic enhancement towards frequency diffusion for video captioning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 37, pp. 3724\u20133732 (2023)","DOI":"10.1609\/aaai.v37i3.25484"},{"key":"4353_CR66","doi-asserted-by":"publisher","first-page":"2367","DOI":"10.1109\/TMM.2023.3295098","volume":"26","author":"S Jing","year":"2023","unstructured":"Jing, S., Zhang, H., Zeng, P., Gao, L., Song, J., Shen, H.T.: Memory-based augmentation network for video captioning. IEEE Trans. Multimedia 26, 2367\u20132379 (2023)","journal-title":"IEEE Trans. Multimedia"},{"key":"4353_CR67","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2023.109906","volume":"145","author":"X Luo","year":"2024","unstructured":"Luo, X., Luo, X., Wang, D., Liu, J., Wan, B., Zhao, L.: Global semantic enhancement network for video captioning. Pattern Recogn. 145, 109906 (2024)","journal-title":"Pattern Recogn."},{"key":"4353_CR68","doi-asserted-by":"crossref","unstructured":"Wang, B., Ma, L., Zhang, W., Jiang, W., Wang, J., Liu, W.: Controllable Video Captioning with POS Sequence Guidance Based on Gated Fusion Network (2019). arXiv:https:\/\/arxiv.org\/abs\/1908.10072","DOI":"10.1109\/ICCV.2019.00273"},{"key":"4353_CR69","doi-asserted-by":"publisher","unstructured":"Zheng, Y., Zhang, Y., Feng, R., Zhang, T., Fan, W.: Stacked multimodal attention network for context-aware video captioning. IEEE Trans. Circuits Syst. Video Technol. 32(1), 31\u201342 (2022). https:\/\/doi.org\/10.1109\/TCSVT.2021.3058626","DOI":"10.1109\/TCSVT.2021.3058626"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-026-04353-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-026-04353-w","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-026-04353-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,13]],"date-time":"2026-03-13T16:28:17Z","timestamp":1773419297000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-026-04353-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2]]},"references-count":69,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2026,2]]}},"alternative-id":["4353"],"URL":"https:\/\/doi.org\/10.1007\/s00371-026-04353-w","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,2]]},"assertion":[{"value":"10 September 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 January 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 February 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"154"}}