{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T10:42:41Z","timestamp":1777632161949,"version":"3.51.4"},"reference-count":60,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2023,11,14]],"date-time":"2023-11-14T00:00:00Z","timestamp":1699920000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,11,14]],"date-time":"2023-11-14T00:00:00Z","timestamp":1699920000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62262009, 61902086"],"award-info":[{"award-number":["62262009, 61902086"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62262009, 61902086"],"award-info":[{"award-number":["62262009, 61902086"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62262009, 61902086"],"award-info":[{"award-number":["62262009, 61902086"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62262009, 61902086"],"award-info":[{"award-number":["62262009, 61902086"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62262009, 61902086"],"award-info":[{"award-number":["62262009, 61902086"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62262009, 61902086"],"award-info":[{"award-number":["62262009, 61902086"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Multimed Info Retr"],"published-print":{"date-parts":[[2023,12]]},"DOI":"10.1007\/s13735-023-00303-7","type":"journal-article","created":{"date-parts":[[2023,11,14]],"date-time":"2023-11-14T03:01:56Z","timestamp":1699930916000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Joint multi-scale information and long-range dependence for video captioning"],"prefix":"10.1007","volume":"12","author":[{"given":"Zhongyi","family":"Zhai","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaofeng","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yishuang","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lingzhong","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Cheng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qian","family":"He","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,11,14]]},"reference":[{"issue":"6","key":"303_CR1","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3355390","volume":"52","author":"N Aafaq","year":"2019","unstructured":"Aafaq N, Mian A, Liu W, Gilani SZ, Shah M (2019) Video description: a survey of methods, datasets, and evaluation metrics. ACM Comput Surv (CSUR) 52(6):1\u201337","journal-title":"ACM Comput Surv (CSUR)"},{"key":"303_CR2","doi-asserted-by":"crossref","unstructured":"Krishnamoorthy N, Malkarnenkar G, Mooney R, Saenko K, Guadarrama S (2013) Generating natural-language video descriptions using text-mined knowledge. In: Twenty-Seventh AAAI conference on artificial intelligence","DOI":"10.1609\/aaai.v27i1.8679"},{"key":"303_CR3","doi-asserted-by":"crossref","unstructured":"Rohrbach M, Qiu W, Titov I, Thater S, Pinkal M, Schiele B (2013) Translating video content to natural language descriptions. In: Proceedings of the IEEE international conference on computer vision, pp 433\u2013440","DOI":"10.1109\/ICCV.2013.61"},{"key":"303_CR4","doi-asserted-by":"crossref","unstructured":"Guadarrama S, Krishnamoorthy N, Malkarnenkar G, Venugopalan S, Mooney R, Darrell T, Saenko K (2013) YouTube2Text: recognizing and describing arbitrary activities using semantic hierarchies and zero-shot recognition. In: Proceedings of the IEEE international conference on computer vision, pp 2712\u20132719","DOI":"10.1109\/ICCV.2013.337"},{"key":"303_CR5","unstructured":"Thomason J, Venugopalan S, Guadarrama S, Saenko K, Mooney R (2014) Integrating language and vision to generate natural language descriptions of videos in the wild. University of Texas at Austin Austin United States, Tech. Rep"},{"key":"303_CR6","doi-asserted-by":"crossref","unstructured":"Xu R, Xiong C, Chen W, Corso J (2015) Jointly modeling deep video and compositional text to bridge vision and language in a unified framework. In: Proceedings of the AAAI conference on artificial intelligence, vol\u00a029(1)","DOI":"10.1609\/aaai.v29i1.9512"},{"issue":"2","key":"303_CR7","doi-asserted-by":"publisher","first-page":"171","DOI":"10.1023\/A:1020346032608","volume":"50","author":"A Kojima","year":"2002","unstructured":"Kojima A, Tamura T, Fukunaga K (2002) Natural language description of human activities from video images based on concept hierarchy of actions. Int J Comput Vis 50(2):171\u2013184","journal-title":"Int J Comput Vis"},{"key":"303_CR8","doi-asserted-by":"crossref","unstructured":"Venugopalan S, Rohrbach M, Donahue J, Mooney R, Darrell T, Saenko K (2015) Sequence to sequence-video to text. In: Proceedings of the IEEE international conference on computer vision, pp 4534\u20134542","DOI":"10.1109\/ICCV.2015.515"},{"key":"303_CR9","doi-asserted-by":"crossref","unstructured":"Yao L, Torabi A, Cho K, Ballas N, Pal C, Larochelle H, Courville A (2015) Describing videos by exploiting temporal structure. In: Proceedings of the IEEE international conference on computer vision, pp 4507\u20134515","DOI":"10.1109\/ICCV.2015.512"},{"key":"303_CR10","doi-asserted-by":"crossref","unstructured":"Pan P, Xu Z, Yang Y, Wu F, Zhuang Y (2016) Hierarchical recurrent neural encoder for video representation with application to captioning. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1029\u20131038","DOI":"10.1109\/CVPR.2016.117"},{"key":"303_CR11","doi-asserted-by":"crossref","unstructured":"Pan Y, Mei T, Yao T, Li H, Rui Y (2016) Jointly modeling embedding and translation to bridge video and language. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4594\u20134602","DOI":"10.1109\/CVPR.2016.497"},{"key":"303_CR12","doi-asserted-by":"crossref","unstructured":"Zhou B, Andonian A, Oliva A, Torralba A (2018) Temporal relational reasoning in videos. In: Proceedings of the European conference on computer vision (ECCV), pp 803\u2013818","DOI":"10.1007\/978-3-030-01246-5_49"},{"key":"303_CR13","unstructured":"Santoro A, Raposo D, Barrett DG, Malinowski M, Pascanu R, Battaglia P, Lillicrap T (2017) A simple neural network module for relational reasoning. In: Advances in neural information processing systems, vol 30"},{"issue":"7","key":"303_CR14","doi-asserted-by":"publisher","first-page":"2631","DOI":"10.1109\/TCYB.2018.2831447","volume":"49","author":"Y Bin","year":"2019","unstructured":"Bin Y, Yang Y, Shen F, Xie N, Shen HT, Li X (2019) Describing video with attention-based bidirectional LSTM. IEEE Trans Cybern 49(7):2631\u20132641","journal-title":"IEEE Trans Cybern"},{"key":"303_CR15","doi-asserted-by":"crossref","unstructured":"Wang X, Girshick R, Gupta A, He K (2018) Non-local neural networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 7794\u20137803","DOI":"10.1109\/CVPR.2018.00813"},{"issue":"5","key":"303_CR16","doi-asserted-by":"publisher","first-page":"502","DOI":"10.1504\/IJCVR.2019.102288","volume":"9","author":"J Lee","year":"2019","unstructured":"Lee J, Kim J (2019) Exploring the effects of non-local blocks on video captioning networks. Int J Comput Vis Robot 9(5):502\u2013514","journal-title":"Int J Comput Vis Robot"},{"key":"303_CR17","unstructured":"Barbu A, Bridge A, Burchill Z, Coroian D, Dickinson S, Fidler S, Michaux A, Mussman S, Narayanaswamy S, Salvi D et\u00a0al (2012) Video in sentences out. arXiv preprint arXiv:1204.2742"},{"key":"303_CR18","doi-asserted-by":"crossref","unstructured":"Khan MUG, Zhang L, Gotoh Y (2011) Human focused video description. In: 2011 IEEE international conference on computer vision workshops (ICCV Workshops). IEEE, pp 1480\u20131487","DOI":"10.1109\/ICCVW.2011.6130425"},{"key":"303_CR19","doi-asserted-by":"crossref","unstructured":"Rafiq G, Rafiq M, Choi GS (2023) Video description: a comprehensive survey of deep learning approaches. Artif Intell Rev, pp 1\u201380","DOI":"10.1007\/s10462-023-10414-6"},{"key":"303_CR20","doi-asserted-by":"crossref","unstructured":"Venugopalan S, Xu H, Donahue J, Rohrbach M, Mooney R, Saenko K (2014) Translating videos to natural language using deep recurrent neural networks. arXiv preprint arXiv:1412.4729","DOI":"10.3115\/v1\/N15-1173"},{"key":"303_CR21","doi-asserted-by":"crossref","unstructured":"Aafaq, N, Akhtar N, Liu W, Gilani SZ, Mian A (2019) Spatio-temporal dynamics and semantic attribute enriched visual encoding for video captioning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 12\u00a0487\u201312\u00a0496","DOI":"10.1109\/CVPR.2019.01277"},{"key":"303_CR22","doi-asserted-by":"crossref","unstructured":"Zhang X, Gao K, Zhang Y, Zhang D, Li J, Tian Q (2017) Task-driven dynamic fusion: reducing ambiguity in video description. In Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3713\u20133721","DOI":"10.1109\/CVPR.2017.662"},{"issue":"9","key":"303_CR23","doi-asserted-by":"publisher","first-page":"2045","DOI":"10.1109\/TMM.2017.2729019","volume":"19","author":"L Gao","year":"2017","unstructured":"Gao L, Guo Z, Zhang H, Xu X, Shen HT (2017) Video captioning with attention-based LSTM and semantic consistency. IEEE Trans Multimed 19(9):2045\u20132055","journal-title":"IEEE Trans Multimed"},{"key":"303_CR24","doi-asserted-by":"crossref","unstructured":"Wang J, Wang W, Huang Y, Wang L, Tan T (2018) M3: multimodal memory modelling for video captioning. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 7512\u20137520","DOI":"10.1109\/CVPR.2018.00784"},{"key":"303_CR25","doi-asserted-by":"publisher","first-page":"9255","DOI":"10.1109\/TPAMI.2021.3132229","volume":"44","author":"F Liu","year":"2021","unstructured":"Liu F, Wu X, You C, Ge S, Zou Y, Sun X (2021) Aligning source visual and target language domains for unpaired video captioning. IEEE Trans Pattern Anal Mach Intell 44:9255\u20139268","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"303_CR26","doi-asserted-by":"crossref","unstructured":"Wang B, Ma L, Zhang W, Liu W (2018) Reconstruction network for video captioning. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 7622\u20137631","DOI":"10.1109\/CVPR.2018.00795"},{"key":"303_CR27","doi-asserted-by":"publisher","first-page":"108332","DOI":"10.1016\/j.asoc.2021.108332","volume":"117","author":"W Ji","year":"2022","unstructured":"Ji W, Wang R, Tian Y, Wang X (2022) An attention based dual learning approach for video captioning. Appl Soft Comput 117:108332","journal-title":"Appl Soft Comput"},{"key":"303_CR28","doi-asserted-by":"publisher","first-page":"202","DOI":"10.1109\/TIP.2021.3120867","volume":"31","author":"L Gao","year":"2021","unstructured":"Gao L, Lei Y, Zeng P, Song J, Wang M, Shen HT (2021) Hierarchical representation network with auxiliary tasks for video captioning and video question answering. IEEE Trans Image Process 31:202\u2013215","journal-title":"IEEE Trans Image Process"},{"key":"303_CR29","doi-asserted-by":"crossref","unstructured":"Seo PH, Nagrani A, Arnab A, Schmid C (2022) End-to-end generative pretraining for multimodal video captioning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 17\u00a0959\u201317\u00a0968","DOI":"10.1109\/CVPR52688.2022.01743"},{"key":"303_CR30","doi-asserted-by":"crossref","unstructured":"Madake J, Bhatlawande S, Purandare S, Shilaskar S, Nikhare Y (2022) Dense video captioning using BiLSTM encoder. In: 3rd international conference for emerging technology (INCET). IEEE, pp 1\u20136","DOI":"10.1109\/INCET54531.2022.9824569"},{"key":"303_CR31","doi-asserted-by":"crossref","unstructured":"Zhou L, Zhou Y, Corso JJ, Socher JJ, Xiong C (2018) End-to-end dense video captioning with masked transformer. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 8739\u20138748","DOI":"10.1109\/CVPR.2018.00911"},{"key":"303_CR32","unstructured":"Chen M, Li Y, Zhang Z, Huang S (2018) TVT: two-view transformer network for video captioning. In: Asian conference on machine learning. PMLR, pp 847\u2013862"},{"key":"303_CR33","doi-asserted-by":"crossref","unstructured":"Zheng Q, Wang C, Tao D (2020) Syntax-aware action targeting for video captioning. In: 2020 IEEE\/CVF conference on computer vision and pattern recognition, CVPR 2020, Seattle, WA, USA, June 13\u201319, 2020. Computer Vision Foundation\/IEEE, pp 13\u00a0093\u201313\u00a0102","DOI":"10.1109\/CVPR42600.2020.01311"},{"key":"303_CR34","doi-asserted-by":"publisher","first-page":"e916","DOI":"10.7717\/peerj-cs.916","volume":"8","author":"H Zhao","year":"2022","unstructured":"Zhao H, Chen Z, Guo L, Han Z (2022) Video captioning based on vision transformer and reinforcement learning. PeerJ Comput Sci 8:e916","journal-title":"PeerJ Comput Sci"},{"key":"303_CR35","doi-asserted-by":"crossref","unstructured":"Xu H, Zeng P, Khan AA (2022) Multimodal interaction fusion network based on transformer for video captioning. In: International symposium on artificial intelligence and robotics. Springer, pp 21\u201336","DOI":"10.1007\/978-981-19-7946-0_3"},{"issue":"1","key":"303_CR36","doi-asserted-by":"publisher","first-page":"229","DOI":"10.1109\/TMM.2019.2924576","volume":"22","author":"C Yan","year":"2019","unstructured":"Yan C, Tu Y, Wang X, Zhang Y, Hao X, Zhang Y, Dai Q (2019) STAT: spatial-temporal attention mechanism for video captioning. IEEE Trans Multimed 22(1):229\u2013241","journal-title":"IEEE Trans Multimed"},{"key":"303_CR37","doi-asserted-by":"crossref","unstructured":"Hu M, Li Y, Fang L, Wang S (2021) A2-FPN: attention aggregation based feature pyramid network for instance segmentation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 15343\u201315352","DOI":"10.1109\/CVPR46437.2021.01509"},{"key":"303_CR38","doi-asserted-by":"crossref","unstructured":"Ghiasi G, Lin T-Y, Le QV (2019) NAS-FPN: learning scalable feature pyramid architecture for object detection. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 7036\u20137045 (2019)","DOI":"10.1109\/CVPR.2019.00720"},{"key":"303_CR39","doi-asserted-by":"crossref","unstructured":"He K, Gkioxari G, Doll\u00e1r P, Girshick R (2017) Mask R-CNN. In: Proceedings of the IEEE international conference on computer vision, pp 2961\u20132969","DOI":"10.1109\/ICCV.2017.322"},{"key":"303_CR40","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"303_CR41","unstructured":"Chung J, Gulcehre C, Cho K, Bengio Y (2014) Empirical evaluation of gated recurrent neural networks on sequence modeling. arXiv preprint arXiv:1412.3555"},{"key":"303_CR42","doi-asserted-by":"crossref","unstructured":"Xu J, Mei T, Yao T, Rui Y (2016) MSR-VTT: a large video description dataset for bridging video and language. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 5288\u20135296","DOI":"10.1109\/CVPR.2016.571"},{"key":"303_CR43","unstructured":"Chen D, Dolan WB (2011) Collecting highly parallel data for paraphrase evaluation. In: Proceedings of the 49th annual meeting of the association for computational linguistics: human language technologies, pp 190\u2013200"},{"key":"303_CR44","doi-asserted-by":"crossref","unstructured":"Hou J, Wu X, Zhao W, Luo J, Jia Y (2019) Joint syntax representation learning and visual cue translation for video captioning. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 8918\u20138927","DOI":"10.1109\/ICCV.2019.00901"},{"key":"303_CR45","doi-asserted-by":"crossref","unstructured":"Papineni K, Roukos S, Ward T, Zhu W (2002) Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th annual meeting of the association for computational linguistics, July 6\u201312, 2002, Philadelphia. ACL, pp 311\u2013318","DOI":"10.3115\/1073083.1073135"},{"key":"303_CR46","unstructured":"Banerjee S, Lavie A (2005) METEOR: an automatic metric for MT evaluation with improved correlation with human judgments. In: Proceedings of the workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization@ACL 2005, Ann Arbor, Michigan, USA, June 29, 2005. Association for Computational Linguistics, pp 65\u201372. [Online]. Available: https:\/\/aclanthology.org\/W05-0909\/"},{"key":"303_CR47","doi-asserted-by":"publisher","unstructured":"Vedantam, R, Zitnick CL, Parikh D (2015) Cider: consensus-based image description evaluation. In: IEEE conference on computer vision and pattern recognition, CVPR 2015, Boston, MA, USA, June 7\u201312, 2015. IEEE Computer Society, pp 4566\u20134575. [Online]. Available: https:\/\/doi.org\/10.1109\/CVPR.2015.7299087","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"303_CR48","unstructured":"Lin C-Y (2004) ROUGE: a package for automatic evaluation of summaries. In: Text Summarization Branches Out. Barcelona, Spain: Association for Computational Linguistics, pp 74\u201381. [Online]. Available: https:\/\/aclanthology.org\/W04-1013"},{"key":"303_CR49","doi-asserted-by":"crossref","unstructured":"Shetty R, Laaksonen J (2016) Frame- and segment-level features and candidate pool evaluation for video caption generation. In: Proceedings of the 2016 ACM conference on multimedia conference, MM 2016, Amsterdam, The Netherlands, October 15\u201319, 2016. ACM, pp 1073\u20131076","DOI":"10.1145\/2964284.2984062"},{"key":"303_CR50","doi-asserted-by":"crossref","unstructured":"Xu J, Yao T, Zhang Y, Mei T (2017) Learning multimodal attention LSTM networks for video captioning. In: Proceedings of the 2017 ACM on multimedia conference, MM 2017, Mountain View, CA, USA, October 23\u201327, 2017. ACM, pp 537\u2013545","DOI":"10.1145\/3123266.3123448"},{"key":"303_CR51","doi-asserted-by":"crossref","unstructured":"Hori C, Hori T, Lee T, Zhang Z, Harsham B, Hershey JR, Marks TK, Sumi K (2017) Attention-based multimodal fusion for video description. In: IEEE international conference on computer vision, ICCV 2017, Venice, Italy, October 22\u201329, 2017. IEEE Computer Society, pp 4203\u20134212","DOI":"10.1109\/ICCV.2017.450"},{"key":"303_CR52","doi-asserted-by":"crossref","unstructured":"Chen Y, Wang S, Zhang W, Huang Q (2018) Less is more: picking informative frames for video captioning, in Computer Vision\u2014ECCV 2018\u201315th European Conference, Munich, Germany, September 8\u201314, (2018) Proceedings, Part XIII, ser. Lecture Notes in Computer Science, vol 11217. Springer, pp 367\u2013384","DOI":"10.1007\/978-3-030-01261-8_22"},{"key":"303_CR53","doi-asserted-by":"crossref","unstructured":"Wang B, Ma L, Zhang W, Jiang W, Wang J, Liu W (2019) Controllable video captioning with POS sequence guidance based on gated fusion network. In: 2019 IEEE\/CVF international conference on computer vision, ICCV 2019, Seoul, Korea (South), October 27\u2013November 2, 2019. IEEE, pp 2641\u20132650","DOI":"10.1109\/ICCV.2019.00273"},{"key":"303_CR54","doi-asserted-by":"crossref","unstructured":"Chen J, Pan Y, Li Y, Yao T, Chao H, Mei T (2019) Temporal deformable convolutional encoder-decoder networks for video captioning. In: The Thirty-Third AAAI conference on artificial intelligence, AAAI 2019, Honolulu, Hawaii, USA, January 27\u2013February 1, AAAI Press, pp 8167\u20138174","DOI":"10.1609\/aaai.v33i01.33018167"},{"key":"303_CR55","unstructured":"Gao L, Li X, Song J, Shen HT (2020) Hierarchical LSTMs with adaptive attention for visual captioning. IEEE Trans Pattern Anal Mach Intell 42(5):1112\u20131131"},{"issue":"2","key":"303_CR56","doi-asserted-by":"publisher","first-page":"880","DOI":"10.1109\/TCSVT.2021.3063423","volume":"32","author":"J Deng","year":"2022","unstructured":"Deng J, Li L, Zhang B, Wang S, Zha Z, Huang Q (2022) Syntax-guided hierarchical attention network for video captioning. IEEE Trans Circuits Syst Video Technol 32(2):880\u2013892","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"issue":"3","key":"303_CR57","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"Russakovsky O, Deng J, Su H, Krause J, Satheesh S, Ma S, Huang Z, Karpathy A, Khosla A, Bernstein MS, Berg AC, Fei-Fei L (2015) Imagenet large scale visual recognition challenge. Int J Comput Vis 115(3):211\u2013252. https:\/\/doi.org\/10.1007\/s11263-015-0816-y","journal-title":"Int J Comput Vis"},{"key":"303_CR58","unstructured":"Kay W, Carreira J, Simonyan K, Zhang B, Hillier C, Vijayanarasimhan S, Viola F, Green T, Back T, Natsev P, Suleyman M, Zisserman A (2017) The kinetics human action video dataset. [Online]. Available: arxiv:1705.06950"},{"key":"303_CR59","unstructured":"Kingma DP, Ba J (2015) Adam: a method for stochastic optimization. In: 3rd International Conference on Learning Representations, ICLR 2015, San Diego, CA, USA, May 7-9, 2015, Conference Track Proceedings. [Online]. Available: arxiv:1412.6980"},{"key":"303_CR60","doi-asserted-by":"crossref","unstructured":"Jin T, Huang S, Chen M, Li Y, Zhang Z (2020) SBAT: video captioning with sparse boundary-aware transformer. arXiv preprint arXiv:2007.11888","DOI":"10.24963\/ijcai.2020\/88"}],"container-title":["International Journal of Multimedia Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-023-00303-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13735-023-00303-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-023-00303-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,12,2]],"date-time":"2023-12-02T14:17:37Z","timestamp":1701526657000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13735-023-00303-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,11,14]]},"references-count":60,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2023,12]]}},"alternative-id":["303"],"URL":"https:\/\/doi.org\/10.1007\/s13735-023-00303-7","relation":{},"ISSN":["2192-6611","2192-662X"],"issn-type":[{"value":"2192-6611","type":"print"},{"value":"2192-662X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,11,14]]},"assertion":[{"value":"10 June 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 September 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 October 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 November 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"All authors disclosed no relevant relationships.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"37"}}