{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,17]],"date-time":"2026-04-17T17:11:40Z","timestamp":1776445900751,"version":"3.51.2"},"reference-count":84,"publisher":"Springer Science and Business Media LLC","issue":"16","license":[{"start":{"date-parts":[[2023,10,28]],"date-time":"2023-10-28T00:00:00Z","timestamp":1698451200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,10,28]],"date-time":"2023-10-28T00:00:00Z","timestamp":1698451200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-023-17328-z","type":"journal-article","created":{"date-parts":[[2023,10,28]],"date-time":"2023-10-28T04:01:36Z","timestamp":1698465696000},"page":"47699-47733","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Video-Captioning Evaluation Metric for Segments (VEMS): A Metric for Segment-level Evaluation of Video Captions with Weighted Frames"],"prefix":"10.1007","volume":"83","author":[{"given":"M.","family":"Ravinder","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Vaidehi","family":"Gupta","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kanishka","family":"Arora","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Arti","family":"Ranjan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5055-3645","authenticated-orcid":false,"given":"Yu-Chen","family":"Hu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,10,28]]},"reference":[{"key":"17328_CR1","doi-asserted-by":"publisher","unstructured":"Xiao H, Xu J, Shi J (2020) Exploring diverse and fine-grained caption for video by incorporating convolutional architecture into LSTM-based model. Pattern Recognit Lett 129:173\u2013180. https:\/\/doi.org\/10.1016\/j.patrec.2019.11.003","DOI":"10.1016\/j.patrec.2019.11.003"},{"key":"17328_CR2","doi-asserted-by":"publisher","unstructured":"Yu H, Wang J, Huang Z, Yang Y, Xu W (2016) Video paragraph captioning using hierarchical recurrent neural networks. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), Las Vegas, pp 4584\u20134593. https:\/\/doi.org\/10.1109\/CVPR.2016.496","DOI":"10.1109\/CVPR.2016.496"},{"key":"17328_CR3","doi-asserted-by":"publisher","unstructured":"Pan J-Y, Hyung-Jeong Yang P, Duygulu, Faloutsos C (2004) Automatic image captioning. In: 2004 IEEE International Conference on Multimedia and Expo (ICME) (IEEE Cat. No.04TH8763), Taipei, pp 1987\u20131990. https:\/\/doi.org\/10.1109\/ICME.2004.1394652","DOI":"10.1109\/ICME.2004.1394652"},{"issue":"9","key":"17328_CR4","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0202789","volume":"13","author":"Y Graham","year":"2018","unstructured":"Graham Y, Awad G, Smeaton A (2018) Evaluation of automatic video captioning using direct assessment. PLoS One 13(9):e0202789. https:\/\/doi.org\/10.1371\/journal.pone.0202789","journal-title":"PLoS One"},{"key":"17328_CR5","doi-asserted-by":"crossref","unstructured":"Pan Y, Yao T, Li H, Mei T (2017) Video captioning with transferred semantic attributes. In :Presented at the Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 65046512. http:\/\/openaccess.thecvf.com\/content_cvpr_2017\/html\/Pan_Video_Captioning_With_CVPR_2017_paper.html. Accessed 11 May 2020","DOI":"10.1109\/CVPR.2017.111"},{"key":"17328_CR6","doi-asserted-by":"crossref","unstructured":"Park J, Song C, Han JH (2017) A study of evaluation metrics and datasets for video captioning. In: ICIIBMS 2017, track 2: Artificial intelligence, robotics and human-computer interaction, Okinawa","DOI":"10.1109\/ICIIBMS.2017.8279760"},{"issue":"10","key":"17328_CR7","doi-asserted-by":"publisher","first-page":"1586","DOI":"10.1007\/s11263-019-01206-z","volume":"127","author":"N Sharif","year":"2019","unstructured":"Sharif N, White L, Bennamoun M, Liu W, Shah SAA (2019) LCEval: Learned composite metric for caption evaluation. Int J Comput Vis 127(10):1586\u20131610. https:\/\/doi.org\/10.1007\/s11263-019-01206-z","journal-title":"Int J Comput Vis"},{"key":"17328_CR8","doi-asserted-by":"publisher","first-page":"172","DOI":"10.1109\/ICIIBMS.2017.8279760","volume-title":"2017 International Conference on Intelligent Informatics and Biomedical Sciences (ICIIBMS)","author":"J Park","year":"2017","unstructured":"Park J, Song C, Han J (2017) A study of evaluation metrics and datasets for video captioning. In: 2017 International Conference on Intelligent Informatics and Biomedical Sciences (ICIIBMS), pp 172\u2013175. https:\/\/doi.org\/10.1109\/ICIIBMS.2017.8279760"},{"key":"17328_CR9","doi-asserted-by":"publisher","first-page":"376","DOI":"10.3115\/v1\/W14-3348","volume-title":"Proceedings of the ninth workshop on statistical machine translation","author":"M Denkowski","year":"2014","unstructured":"Denkowski M, Lavie A (2014) Meteor Universal: Language specific translation evaluation for any target language. In: Proceedings of the ninth workshop on statistical machine translation. Baltimore, Maryland, pp 376\u2013380. https:\/\/doi.org\/10.3115\/v1\/W14-3348"},{"key":"17328_CR10","doi-asserted-by":"crossref","unstructured":"Vedantam R, Zitnick CL, Parikh D (2015) CIDEr: Consensus-based image description evaluation. ArXiv14115726 Cs. http:\/\/arxiv.org\/abs\/1411.5726. Accessed 9 May 2020","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"17328_CR11","doi-asserted-by":"publisher","unstructured":"Papineni K, Roukos S, Ward T, Zhu W-J (2002) Bleu: A method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, Philadelphia, pp 311\u2013318. https:\/\/doi.org\/10.3115\/1073083.1073135","DOI":"10.3115\/1073083.1073135"},{"key":"17328_CR12","first-page":"65","volume-title":"Proceedings of the ACL workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization","author":"S Banerjee","year":"2005","unstructured":"Banerjee S, Lavie A (2005) METEOR: An automatic metric for mt evaluation with improved correlation with human judgments. In: Proceedings of the ACL workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization, pp 65\u201372"},{"key":"17328_CR13","doi-asserted-by":"publisher","unstructured":"Park J, Song C, Han J (2017) A study of evaluation metrics and datasets for video captioning. In: 2017 International Conference on Intelligent Informatics and Biomedical Sciences (ICIIBMS), pp 172\u2013175. https:\/\/doi.org\/10.1109\/ICIIBMS.2017.8279760","DOI":"10.1109\/ICIIBMS.2017.8279760"},{"key":"17328_CR14","doi-asserted-by":"publisher","unstructured":"Anderson P, Fernando B, Johnson M, Gould S (2016) SPICE: Semantic propositional image caption evaluation. In: Computer vision \u2013 ECCV 2016, Cham, pp 382\u2013398. https:\/\/doi.org\/10.1007\/978-3-319-46454-1_24","DOI":"10.1007\/978-3-319-46454-1_24"},{"key":"17328_CR15","doi-asserted-by":"publisher","unstructured":"Aafaq N, Mian A, Liu W, Gilani SJ, Shah M (2020) Video description: A survey of methods, datasets and evaluation metrics. ACM Comput Surv 52(6):1\u201337. https:\/\/doi.org\/10.1145\/3355390","DOI":"10.1145\/3355390"},{"issue":"1","key":"17328_CR16","doi-asserted-by":"publisher","first-page":"51","DOI":"10.1002\/aris.1440370103","volume":"37","author":"GG Chowdhury","year":"2003","unstructured":"Chowdhury GG (2003) Natural language processing. Annu Rev Inform Sci Technol 37(1):51\u201389. https:\/\/doi.org\/10.1002\/aris.1440370103","journal-title":"Annu Rev Inform Sci Technol"},{"issue":"76","key":"17328_CR17","first-page":"2493","volume":"12","author":"R Collobert","year":"2011","unstructured":"Collobert R, Weston J, Bottou L, Karlen M, Kavukcuoglu K, Kuksa P (2011) Natural language processing (almost) from Scratch. J Mach Learn Res 12(76):2493\u20132537","journal-title":"J Mach Learn Res"},{"key":"17328_CR18","volume-title":"Natural language understanding","author":"J Allen","year":"1995","unstructured":"Allen J (1995) Natural language understanding. Pearson"},{"key":"17328_CR19","doi-asserted-by":"crossref","unstructured":"Corley C, Mihalcea R (2005) Measuring the semantic similarity of texts. In: Proceedings of the ACL Workshop on Empirical Modeling of Semantic Equivalence and Entailment, Ann Arbor, pp 13\u201318. https:\/\/www.aclweb.org\/anthology\/W05-1203. Accessed 21 May 2020","DOI":"10.3115\/1631862.1631865"},{"key":"17328_CR20","volume-title":"A review of semantic similarity measures in wordNet 1","author":"L Meng","year":"2013","unstructured":"Meng L, Huang R, Gu J (2013) A review of semantic similarity measures in wordNet 1"},{"key":"17328_CR21","doi-asserted-by":"publisher","DOI":"10.1017\/atsip.2019.26","volume-title":"APSIPA Transactions on Signal Information Processing","author":"W Zeng","year":"2020","unstructured":"Zeng W (2020) Toward human-centric deep video understanding. In: APSIPA Transactions on Signal Information Processing, vol 9. https:\/\/doi.org\/10.1017\/atsip.2019.26"},{"key":"17328_CR22","doi-asserted-by":"publisher","first-page":"1073","DOI":"10.1145\/2964284.2984062","volume-title":"Proceedings of the 24th ACM international conference on Multimedia","author":"R Shetty","year":"2016","unstructured":"Shetty R, Laaksonen J (2016) Frame- and segment-level features and candidate pool evaluation for video caption generation. In: Proceedings of the 24th ACM international conference on Multimedia, pp 1073\u20131076. https:\/\/doi.org\/10.1145\/2964284.2984062"},{"key":"17328_CR23","unstructured":"Lebron L et al (2022) Bertha: Video captioning evaluation via transfer-learned human assessment. arXiv preprint arXiv:2201.10243"},{"key":"17328_CR24","unstructured":"Chen H, Li J, Hu X (2020) Delving deeper into the decoder for video captioning. ArXiv200105614 Cs. http:\/\/arxiv.org\/abs\/2001.05614. Accessed 6 Feb 2020"},{"key":"17328_CR25","doi-asserted-by":"crossref","unstructured":"Cui Y, Yang G, Veit A, Huang X, Belongie S (2018) Learning to evaluate image captioning. ArXiv180606422 Cs. http:\/\/arxiv.org\/abs\/1806.06422. Accessed 9 May 2020","DOI":"10.1109\/CVPR.2018.00608"},{"key":"17328_CR26","doi-asserted-by":"publisher","unstructured":"Hodosh M, Hockenmaier J (2016) Focused evaluation for image description with binary forced-choice tasks. In: Proceedings of the 5th Workshop on Vision and Language, Berlin, 28, p 19. https:\/\/doi.org\/10.18653\/v1\/W16-3203","DOI":"10.18653\/v1\/W16-3203"},{"key":"17328_CR27","volume-title":"MSR-VTT: A large video description dataset for bridging video and language","author":"J Xu","year":"2016","unstructured":"Xu J, Mei T, Yao T, Rui Y (2016) MSR-VTT: A large video description dataset for bridging video and language. https:\/\/www.microsoft.com\/en-us\/research\/publication\/msr-vtt-a-large-video-description-dataset-for-bridging-video-and-language\/. Accessed 21 May 2020"},{"key":"17328_CR28","unstructured":"\u201cM-VAD,\u201d Mila. https:\/\/mila.quebec\/en\/publications\/public-datasets\/m-vad\/. Accessed 21 May 2020"},{"key":"17328_CR29","first-page":"1916","volume-title":"Presented at the Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"Z Shen","year":"2017","unstructured":"Shen Z et al (2017) Weakly supervised dense video captioning. In: Presented at the Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 1916\u20131924. http:\/\/openaccess.thecvf.com\/content_cvpr_2017\/html\/Shen_Weakly_Supervised_Dense_CVPR_2017_paper.html. Accessed 21 May 2020"},{"key":"17328_CR30","doi-asserted-by":"publisher","unstructured":"Gao L, Guo Z, Zhang H, Xu X, Shen HT (2017) Video captioning with attention-based LSTM and semantic consistency. IEEE Trans Multimedia 19(9):2045\u20132055. https:\/\/doi.org\/10.1109\/TMM.2017.2729019","DOI":"10.1109\/TMM.2017.2729019"},{"key":"17328_CR31","doi-asserted-by":"crossref","unstructured":"Chen Y, Wang S, Zhang W, Huang Q (2018) Less is more: Picking informative frames for video captioning. ArXiv180301457 Cs. http:\/\/arxiv.org\/abs\/1803.01457. Accessed 9 May 2020","DOI":"10.1007\/978-3-030-01261-8_22"},{"key":"17328_CR32","unstructured":"Chen D, Dolan W (2011) Collecting highly parallel data for paraphrase evaluation. In: Proceedings of the 49th Annual Meeting of the Association for Computational Linguistics: Human Language Technologies, Portland, pp 190\u2013200. https:\/\/www.aclweb.org\/anthology\/P11-1020. Accessed 9 May 2020"},{"key":"17328_CR33","unstructured":"https:\/\/github.com\/Naomi98\/VEMS-Metric\/blob\/master\/Data\/MSVD-S.csv"},{"key":"17328_CR34","doi-asserted-by":"publisher","unstructured":"Munk MM, Voz\u00e1r M (2013) Data pre-processing evaluation for text mining: Transaction\/sequence model. Procedia Comput Sci 18:1198\u20131207. https:\/\/doi.org\/10.1016\/j.procs.2013.05.286","DOI":"10.1016\/j.procs.2013.05.286"},{"key":"17328_CR35","unstructured":"Aafaq N, Mian A, Liu W, Gilani SZ, Shah M (2020) arXiv:1806.00186v4 [cs.CV]"},{"key":"17328_CR36","volume-title":"Translating video content to natural language descriptions","author":"M Rohrbach","year":"2013","unstructured":"Rohrbach M, Qiu W, Titov I, Thater S, Pinkal M, Schiele B (2013) Translating video content to natural language descriptions. ICCV"},{"key":"17328_CR37","doi-asserted-by":"publisher","unstructured":"Kondrak G (2005) N-gram similarity and distance. In: String processing and information retrieval, Berlin, Heidelberg, pp 115\u2013126. https:\/\/doi.org\/10.1007\/11575832_13","DOI":"10.1007\/11575832_13"},{"key":"17328_CR38","doi-asserted-by":"publisher","unstructured":"Aizawa A (2003) An information-theoretic perspective of TF\u2013IDF measures. Inf Process Manag 39(1):45\u201365. https:\/\/doi.org\/10.1016\/S0306-4573(02)00021-3.","DOI":"10.1016\/S0306-4573(02)00021-3"},{"issue":"4","key":"17328_CR39","doi-asserted-by":"publisher","first-page":"309","DOI":"10.1147\/rd.14.0309","volume":"1","author":"HP Luhn","year":"1957","unstructured":"Luhn HP (1957) A statistical approach to mechanized encoding and searching of literary information. IBM J Res Dev 1(4):309\u2013317. https:\/\/doi.org\/10.1147\/rd.14.0309","journal-title":"IBM J Res Dev"},{"key":"17328_CR40","doi-asserted-by":"publisher","unstructured":"Sp\u00e4rck Jones K (1972) A statistical interpretation of term specificity and its application in retrieval. J Doc 28:11\u201321. https:\/\/doi.org\/10.1108\/eb026526","DOI":"10.1108\/eb026526"},{"key":"17328_CR41","volume-title":"The European Conference on Computer Vision (ECCV)","author":"N Dalal","year":"2006","unstructured":"Dalal N, Triggs B, Schmid C (2006) Human detection using oriented histograms of flow and appearance. In: The European Conference on Computer Vision (ECCV)"},{"key":"17328_CR42","volume-title":"2013 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","author":"P Das","year":"2013","unstructured":"Das P, Xu C, Doell RF, Corso JJ (2013) A thousand frames in just a few words: Lingual description of videos through latent topics and sparse object stitching. In: 2013 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"17328_CR43","volume-title":"2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","author":"A Das","year":"2017","unstructured":"Das A, Kottur S, Gupta K, Singh A, Yadav D, Moura JMF, Parikh D, Batra D (2017) Visual dialog. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"17328_CR44","first-page":"2","volume":"186","author":"J Deng","year":"2009","unstructured":"Deng J, Li K, Do M, Su H, Fei-Fei L (2009) Construction and analysis of a large scale image ontology. Vision Sciences Society 186:2","journal-title":"Vision Sciences Society"},{"key":"17328_CR45","volume-title":"2nd ACM International Conference on Multimedia Retrieval (ICMR)","author":"D Ding","year":"2012","unstructured":"Ding D, Metze F, Rawat S, Schulam PF, Burger S, Younessian E, Bao L, Christel MG, Hauptmann A (2012) Beyond audio and video retrieval: towards multimedia summarization. In: 2nd ACM International Conference on Multimedia Retrieval (ICMR)"},{"key":"17328_CR46","volume-title":"NAACL-HTL Translating videos to natural language using deep recurrent neural networks","author":"S Venugopalan","year":"2015","unstructured":"Venugopalan S, Xu H, Donahue J, Rohrbach M, Mooney R, Saenko K (2015) NAACL-HTL Translating videos to natural language using deep recurrent neural networks."},{"key":"17328_CR47","unstructured":"https:\/\/github.com\/Naomi98\/VEMS-Metric\/tree\/master\/Code"},{"issue":"9","key":"17328_CR48","doi-asserted-by":"crossref","first-page":"1627","DOI":"10.1109\/TPAMI.2009.167","volume":"32","author":"PF Felzenszwalb","year":"2010","unstructured":"Felzenszwalb PF, Girshick RB, McAllester D, Ramanan D (2010) Object detection with discriminatively trained part-based models. IEEE Trans Pattern Anal Mach Intell 32(9):1627\u20131645","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"17328_CR49","volume-title":"2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Z Gan","year":"2017","unstructured":"Gan Z, Gan C, He X, Pu Y, Tran K, Gao J, Carin L, Deng L (2017) Semantic compositional networks for visual captioning. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"17328_CR50","volume-title":"Proceedings of the 17th annual TREC Video Retrieval Evaluation: TRECVID 2017 Workshop","author":"A George","year":"2017","unstructured":"George A, Asad B, Jonathan F, David J, Andrew D, Willie M, Martial M, Alan S, Yvette G, Wessel K (2017) TRECVID 2017: Evaluating ad-hoc and instance video search, events detection, video captioning, and hyperlinking. In: Proceedings of the 17th annual TREC Video Retrieval Evaluation: TRECVID 2017 Workshop"},{"key":"17328_CR51","first-page":"1082","volume-title":"Proceedings of the 2016 ACM on Multimedia Conference","author":"J Dong","year":"2016","unstructured":"Dong J, Li X, Lan W, Huo Y, Snoek CGM (2016) Early embedding and late reranking for video captioning. In: Proceedings of the 2016 ACM on Multimedia Conference. ACM, pp 1082\u20131086"},{"key":"17328_CR52","doi-asserted-by":"crossref","unstructured":"Elliott D, Keller F (2014) Comparing automatic evaluation measures for image description. In: Proceedings of the 52nd annual meeting of the association for computational linguistics: Short papers, pp 452\u2013457","DOI":"10.3115\/v1\/P14-2074"},{"issue":"2","key":"17328_CR53","doi-asserted-by":"crossref","first-page":"303","DOI":"10.1007\/s11263-009-0275-4","volume":"88","author":"M Everingham","year":"2010","unstructured":"Everingham M, Gool LV, Williams CKI, Winn J, Zisserman A (2010) The pascal visual object classes (VOC) challenge. Int J Comput Vis 88(2):303\u2013338","journal-title":"Int J Comput Vis"},{"key":"17328_CR54","volume-title":"2015 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","author":"H Fang","year":"2015","unstructured":"Fang H, Gupta S, Iandola F, Srivastava RK, Deng L, Dollr P, Gao J, He X, Mitchell M, Platt JC et al (2015) From captions to visual concepts and back. In: 2015 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"17328_CR55","volume-title":"The European Conference on Computer Vision (ECCV)","author":"A Farhadi","year":"2010","unstructured":"Farhadi A, Hejrati M, Sadeghi MA, Young P, Rashtchian C, Hockenmaier J, Forsyth D (2010) Every picture tells a story: Generating sentences from images. In: The European Conference on Computer Vision (ECCV)"},{"key":"17328_CR56","doi-asserted-by":"crossref","DOI":"10.7551\/mitpress\/7287.001.0001","volume-title":"WordNet","author":"C Fellbaum","year":"1998","unstructured":"Fellbaum C (1998) WordNet. Wiley Online Library"},{"key":"17328_CR57","volume-title":"2008 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","author":"P Felzenszwalb","year":"2008","unstructured":"Felzenszwalb P, McAllester D, Ramanan D (2008) A discriminatively trained, multiscale, deformable part model. In: 2008 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"17328_CR58","volume-title":"2010 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","author":"PF Felzenszwalb","year":"2010","unstructured":"Felzenszwalb PF, Girshick RB, McAllester D (2010) Cascade object detection with deformable part models. In: 2010 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"17328_CR59","unstructured":"Ghanem B, Niebles J, Snoek C, Heilbron F, Alwassel H, Khrisna R, Escorcia V, Hata K, Buch S (2017) ActivityNet challenge 2017 summary. arXiv preprint arXiv:1710.08011"},{"key":"17328_CR60","doi-asserted-by":"crossref","unstructured":"Gella S, Lewis M, Rohrbach M (2018) A dataset for telling the stories of social media videos. In: Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing, p 968974","DOI":"10.18653\/v1\/D18-1117"},{"key":"17328_CR61","volume-title":"IEEE ICCV","author":"S Gong","year":"2003","unstructured":"Gong S, Xiang T (2003) Recognition of group activities using dynamic probabilistic networks. In: IEEE ICCV"},{"key":"17328_CR62","volume-title":"Proceedings of the 24th ACM international conference on Multimedia","author":"J Dong","year":"2016","unstructured":"Dong J et al (2016) Early embedding and late reranking for video captioning. In: Proceedings of the 24th ACM international conference on Multimedia"},{"issue":"1","key":"17328_CR63","doi-asserted-by":"crossref","first-page":"3","DOI":"10.1017\/S1351324915000339","volume":"23","author":"Y Graham","year":"2017","unstructured":"Graham Y, Baldwin T, Moffat A, Zobel J (2017) Can machine translation systems be evaluated by the crowd alone. Nat Lang Eng 23(1):3\u201330","journal-title":"Nat Lang Eng"},{"key":"17328_CR64","first-page":"1764","volume-title":"Proceedings of the 31st International Conference on Machine Learning (ICML-14)","author":"A Graves","year":"2014","unstructured":"Graves A, Jaitly N (2014) Towards end-to-end speech recognition with recurrent neural networks. In: Proceedings of the 31st International Conference on Machine Learning (ICML-14), pp 1764\u20131772"},{"key":"17328_CR65","doi-asserted-by":"crossref","unstructured":"Graves A, Mohamed A, Hinton G (2013) Speech recognition with deep recurrent neural networks. In: IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp 6645-6649","DOI":"10.1109\/ICASSP.2013.6638947"},{"key":"17328_CR66","volume-title":"IEEE ICCV","author":"S Guadarrama","year":"2013","unstructured":"Guadarrama S, Krishnamoorthy N, Malkarnenkar G, Venugopalan S, Mooney R, Darrell T, Saenko K (2013) Recognizing and describing activities using semantic hierarchies and zero-shot recognition. IEEE ICCV"},{"key":"17328_CR67","first-page":"1640","volume-title":"Intelligent Robots and Systems (IROS)","author":"S Guadarrama","year":"2013","unstructured":"Guadarrama S, Riano L, Golland D, Go D, Jia Y, Klein D, Abbeel P, Darrell T et al (2013) Grounding spatial relations for human-robot interaction. In: Intelligent Robots and Systems (IROS), pp 1640\u20131647"},{"key":"17328_CR68","unstructured":"Hakeem A, Sheikh Y, Shah M (2004) CASEE: A hierarchical event representation for the analysis of videos. In: AAAI, pp 263\u2013268"},{"key":"17328_CR69","first-page":"44","volume-title":"Second joint conference on lexical and computational semantics (* SEM), volume 1: Proceedings of the main conference and the shared task: Semantic textual similarity","author":"L Han","year":"2013","unstructured":"Han L, Kashyap AL, Finin T, Mayfield J, Weese J (2013) UMBC-EBIQUITY-CORE: Semantic textual similarity systems. In: Second joint conference on lexical and computational semantics (* SEM), volume 1: Proceedings of the main conference and the shared task: Semantic textual similarity, vol 1, pp 44\u201352"},{"key":"17328_CR70","volume-title":"The European Conference on Computer Vision (ECCV)","author":"P Hanckmann","year":"2012","unstructured":"Hanckmann P, Schutte K, Burghouts GJ (2012) Automated textual descriptions for a wide range of video events with 48 human actions. In: The European Conference on Computer Vision (ECCV)"},{"key":"17328_CR71","volume-title":"The European Conference on Computer Vision (ECCV)","author":"D Harwath","year":"2018","unstructured":"Harwath D, Recasens A, Suris D, Chuang G, Torralba A, Glass J (2018) Jointly discovering visual objects and spoken words from raw sensory input. In: The European Conference on Computer Vision (ECCV)"},{"key":"17328_CR72","volume-title":"2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","author":"K He","year":"2016","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)"},{"issue":"8","key":"17328_CR73","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter S, Schmidhuber J (1997) Long short-term memory. Neural Comput 9(8):1735\u20131780","journal-title":"Neural Comput"},{"key":"17328_CR74","first-page":"164","volume-title":"Proceedings of 15th International Conference on Pattern Recognition","author":"S Hongeng","year":"2000","unstructured":"Hongeng S, Brmond F, Nevatia R (2000) Bayesian framework for video surveillance application. In: Proceedings of 15th International Conference on Pattern Recognition, vol 1, pp 164\u2013170"},{"key":"17328_CR75","volume-title":"ICLR","author":"DA Hudson","year":"2018","unstructured":"Hudson DA, Manning CD (2018) Compositional attention networks for machine reasoning. In: ICLR"},{"key":"17328_CR76","doi-asserted-by":"crossref","first-page":"675","DOI":"10.1145\/2647868.2654889","volume-title":"Proceedings of the 22nd ACM International Conference on Multimedia","author":"Y Jia","year":"2014","unstructured":"Jia Y, Shelhamer E, Donahue J, Karayev S, Long J, Girshick R, Guadarrama S, Darrell T (2014) Caffe: Convolutional architecture for fast feature embedding. In: Proceedings of the 22nd ACM International Conference on Multimedia. ACM, pp 675\u2013678"},{"key":"17328_CR77","doi-asserted-by":"crossref","unstructured":"Jin Q, Chen J, Chen S, Xiong Y, Hauptmann A (2016) Describing videos using multi-modal fusion. In: Proceedings of the 2016 ACM on Multimedia Conference. ACM, pp 1087\u20131091","DOI":"10.1145\/2964284.2984065"},{"key":"17328_CR78","doi-asserted-by":"crossref","unstructured":"Johnson J, Hariharan B, Maaten LVD, Fei-Fei L, Zitnick CL, Girshick R (2017) CLEVR: A diagnostic dataset for compositional language and elementary visual reasoning. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","DOI":"10.1109\/CVPR.2017.215"},{"key":"17328_CR79","first-page":"27","volume-title":"Workshop on innovative hybrid approaches to the processing of textual data","author":"MUG Khan","year":"2012","unstructured":"Khan MUG, Gotoh Y (2012) Describing video contents in natural language. In: Workshop on innovative hybrid approaches to the processing of textual data. Association for Computational Linguistics, pp 27\u201335"},{"key":"17328_CR80","volume-title":"IEEE International Conference on Computer Vision Workshops (ICCV Workshops)","author":"MUG Khan","year":"2011","unstructured":"Khan MUG, Zhang L, Gotoh Y (2011) Human focused video description. In: IEEE International Conference on Computer Vision Workshops (ICCV Workshops)"},{"key":"17328_CR81","doi-asserted-by":"crossref","unstructured":"Kilickaya M, Erdem A, Ikizler-Cinbis N, Erdem E (2016) Reevaluating automatic metrics for image captioning. arXiv preprint arXiv:1612.07600","DOI":"10.18653\/v1\/E17-1019"},{"key":"17328_CR82","volume-title":"The European Conference on Computer Vision (ECCV)","author":"J Kim","year":"2018","unstructured":"Kim J, Rohrbach A, Darrell T, Canny J, Akata Z (2018) Textual explanations for self-driving vehicles. In: The European Conference on Computer Vision (ECCV)"},{"key":"17328_CR83","first-page":"1682","volume-title":"Advances in neural information processing systems","author":"M Malinowski","year":"2014","unstructured":"Malinowski M, Fritz M (2014) A multi-world approach to question answering about real-world scenes based on uncertain input. In: Advances in neural information processing systems, pp 1682\u20131690"},{"key":"17328_CR84","unstructured":"Langkilde-Geary I, Knight K. Halogen input representation"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-17328-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-023-17328-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-17328-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,5,7]],"date-time":"2024-05-07T11:21:29Z","timestamp":1715080889000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-023-17328-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,28]]},"references-count":84,"journal-issue":{"issue":"16","published-online":{"date-parts":[[2024,5]]}},"alternative-id":["17328"],"URL":"https:\/\/doi.org\/10.1007\/s11042-023-17328-z","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,10,28]]},"assertion":[{"value":"28 February 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 September 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 September 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 October 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they do not have any conflict of interests that influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of Interest"}},{"value":"No animals were involved in this study. All applicable international, national, and\/or institutional guidelines for the care and use of animals were followed obtained from the corresponding author upon written request.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}}]}}