{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,25]],"date-time":"2025-12-25T07:24:49Z","timestamp":1766647489083},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"30","license":[{"start":{"date-parts":[[2024,2,14]],"date-time":"2024-02-14T00:00:00Z","timestamp":1707868800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,2,14]],"date-time":"2024-02-14T00:00:00Z","timestamp":1707868800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-023-17904-3","type":"journal-article","created":{"date-parts":[[2024,2,14]],"date-time":"2024-02-14T07:02:18Z","timestamp":1707894138000},"page":"75195-75216","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Leveraging facial expressions as emotional context in image captioning"],"prefix":"10.1007","volume":"83","author":[{"given":"Riju","family":"Das","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nan","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Soumyabrata","family":"Dev","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,2,14]]},"reference":[{"key":"17904_CR1","doi-asserted-by":"publisher","first-page":"291","DOI":"10.1016\/j.neucom.2018.05.080","volume":"311","author":"S Bai","year":"2018","unstructured":"Bai S, An S (2018) A survey on automatic image caption generation. Neurocomputing 311:291\u2013304","journal-title":"Neurocomputing"},{"issue":"200","key":"17904_CR2","first-page":"039","volume":"4","author":"S Batra","year":"2022","unstructured":"Batra S, Wang H, Nag A et al (2022) DMCNet: Diversified model combination network for understanding engagement from video screengrabs. Systems and Soft Computing 4(200):039","journal-title":"Systems and Soft Computing"},{"key":"17904_CR3","unstructured":"Bradski G, Kaehler A (2008) Learning OpenCV: Computer vision with the OpenCV library. \" O\u2019Reilly Media, Inc.\""},{"key":"17904_CR4","first-page":"1","volume":"2020","author":"Y Chu","year":"2020","unstructured":"Chu Y, Yue X, Yu L et al (2020) Automatic image captioning based on resnet50 and lstm with soft attention. Wirel Commun Mob Comput 2020:1\u20137","journal-title":"Wirel Commun Mob Comput"},{"key":"17904_CR5","first-page":"13","volume":"12","author":"A Cowen","year":"2018","unstructured":"Cowen A (2018) How many different kinds of emotion are there? Age 12:13","journal-title":"Age"},{"key":"17904_CR6","doi-asserted-by":"crossref","unstructured":"De Marneffe MC, Manning CD (2008) Stanford typed dependencies manual. Technical report, Stanford University, Tech. rep","DOI":"10.3115\/1608858.1608859"},{"key":"17904_CR7","doi-asserted-by":"crossref","unstructured":"Deng J, Guo J, Ververas E, et\u00a0al (2020) Retinaface: Single-shot multi-level face localisation in the wild. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. pp 5203\u20135212","DOI":"10.1109\/CVPR42600.2020.00525"},{"issue":"2","key":"17904_CR8","doi-asserted-by":"publisher","first-page":"124","DOI":"10.1037\/h0030377","volume":"17","author":"P Ekman","year":"1971","unstructured":"Ekman P, Friesen WV (1971) Constants across cultures in the face and emotion. J Pers Soc Psychol 17(2):124","journal-title":"J Pers Soc Psychol"},{"issue":"3875","key":"17904_CR9","doi-asserted-by":"publisher","first-page":"86","DOI":"10.1126\/science.164.3875.86","volume":"164","author":"P Ekman","year":"1969","unstructured":"Ekman P, Sorenson ER, Friesen WV (1969) Pan-cultural elements in facial displays of emotion. Science 164(3875):86\u201388","journal-title":"Science"},{"key":"17904_CR10","doi-asserted-by":"crossref","unstructured":"Farhadi A, Hejrati M, Sadeghi MA, et\u00a0al (2010) Every picture tells a story: Generating sentences from images. In: European conference on computer vision. Springer, pp 15\u201329","DOI":"10.1007\/978-3-642-15561-1_2"},{"key":"17904_CR11","doi-asserted-by":"crossref","unstructured":"Gan Z, Gan C, He X, et\u00a0al (2017) Semantic compositional networks for visual captioning. In: Proceedings of the IEEE conference on computer vision and pattern recognition. pp 5630\u20135639","DOI":"10.1109\/CVPR.2017.127"},{"key":"17904_CR12","doi-asserted-by":"publisher","first-page":"26,391","DOI":"10.1109\/ACCESS.2018.2831927","volume":"6","author":"J Guo","year":"2018","unstructured":"Guo J, Lei Z, Wan J et al (2018) Dominant and complementary emotion recognition from still images of faces. IEEE Access 6:26,391-26,403","journal-title":"IEEE Access"},{"key":"17904_CR13","doi-asserted-by":"publisher","first-page":"853","DOI":"10.1613\/jair.3994","volume":"47","author":"M Hodosh","year":"2013","unstructured":"Hodosh M, Young P, Hockenmaier J (2013) Framing image description as a ranking task: Data, models and evaluation metrics. J Artif Intell Res 47:853\u2013899","journal-title":"J Artif Intell Res"},{"key":"17904_CR14","doi-asserted-by":"publisher","first-page":"101","DOI":"10.1016\/j.patrec.2018.04.010","volume":"115","author":"N Jain","year":"2018","unstructured":"Jain N, Kumar S, Kumar A et al (2018) Hybrid deep neural networks for face emotion recognition. Pattern Recogn Lett 115:101\u2013106","journal-title":"Pattern Recogn Lett"},{"key":"17904_CR15","doi-asserted-by":"crossref","unstructured":"Johnson J, Karpathy A, Fei-Fei L (2016) Densecap: Fully convolutional localization networks for dense captioning. In: Proceedings of the IEEE conference on computer vision and pattern recognition. pp 4565\u20134574","DOI":"10.1109\/CVPR.2016.494"},{"key":"17904_CR16","doi-asserted-by":"crossref","unstructured":"Karpathy A, Fei-Fei L (2015) Deep visual-semantic alignments for generating image descriptions. In: Proceedings of the IEEE conference on computer vision and pattern recognition. pp 3128\u20133137","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"17904_CR17","doi-asserted-by":"crossref","unstructured":"Kim DJ, Choi J, Oh TH, et\u00a0al (2019) Dense relational captioning: Triple-stream networks for relationship-based captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp 6271\u20136280","DOI":"10.1109\/CVPR.2019.00643"},{"issue":"11","key":"17904_CR18","doi-asserted-by":"publisher","first-page":"7348","DOI":"10.1109\/TPAMI.2021.3119754","volume":"44","author":"DJ Kim","year":"2021","unstructured":"Kim DJ, Oh TH, Choi J et al (2021) Dense relational image captioning via multi-task triple-stream networks. IEEE Trans Pattern Anal Mach Intell 44(11):7348\u20137362","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"17904_CR19","doi-asserted-by":"crossref","unstructured":"Liang PP, Liu Z, Zadeh A, et\u00a0al (2018) Multimodal language analysis with recurrent multistage fusion. arXiv preprint arXiv:1808.03920","DOI":"10.18653\/v1\/D18-1014"},{"key":"17904_CR20","doi-asserted-by":"publisher","first-page":"677","DOI":"10.1007\/s11042-017-5532-x","volume":"78","author":"AA Liu","year":"2019","unstructured":"Liu AA, Shao Z, Wong Y et al (2019) Lstm-based multi-label video event detection. Multimed Tools Appl 78:677\u2013695","journal-title":"Multimed Tools Appl"},{"key":"17904_CR21","unstructured":"Lu J, Yang J, Batra D et al (2016) Hierarchical question-image co-attention for visual question answering. Adv Neural Inf Process Syst 29"},{"issue":"5\u20136","key":"17904_CR22","doi-asserted-by":"publisher","first-page":"555","DOI":"10.1016\/S0893-6080(03)00115-1","volume":"16","author":"M Matsugu","year":"2003","unstructured":"Matsugu M, Mori K, Mitari Y et al (2003) Subject independent facial expression recognition with robust face detection using a convolutional neural network. Neural Netw 16(5\u20136):555\u2013559","journal-title":"Neural Netw"},{"key":"17904_CR23","doi-asserted-by":"crossref","unstructured":"Mohamad\u00a0Nezami O, Dras M, Anderson P, et\u00a0al (2018) Face-cap: Image captioning using facial expression analysis. In: Joint European conference on machine learning and knowledge discovery in databases. Springer, pp 226\u2013240","DOI":"10.1007\/978-3-030-10925-7_14"},{"key":"17904_CR24","doi-asserted-by":"crossref","unstructured":"Pi\u00f3rkowska M, Wrobel M (2017) Basic emotions. Encyclopedia of personality and individual differences. pp 1\u20136","DOI":"10.1007\/978-3-319-28099-8_495-1"},{"key":"17904_CR25","unstructured":"Pramerdorfer C, Kampel M (2016) Facial expression recognition using convolutional neural networks: state of the art. arXiv:1612.02903"},{"key":"17904_CR26","unstructured":"Realm H (2022) Image caption generator using python | flickr dataset | deep learning tutorial. https:\/\/www.hackersrealm.net\/post\/image-caption-generator-using-python"},{"key":"17904_CR27","doi-asserted-by":"publisher","unstructured":"Serengil SI, Ozpinar A (2021) Hyperextended lightface: A facial attribute analysis framework. In: 2021 International Conference on Engineering and Emerging Technologies (ICEET). IEEE, pp 1\u20134 https:\/\/doi.org\/10.1109\/ICEET53442.2021.9659697","DOI":"10.1109\/ICEET53442.2021.9659697"},{"key":"17904_CR28","unstructured":"Shao Z, Han J, Marnerides D, et\u00a0al (2022) Region-object relation-aware dense captioning via transformer. IEEE Trans Neur Netw Learn Syst"},{"key":"17904_CR29","doi-asserted-by":"crossref","unstructured":"Shao Z, Han J, Debattista K, et\u00a0al (2023) Textual context-aware dense captioning with diverse words. IEEE Trans Multimed","DOI":"10.1109\/TMM.2023.3241517"},{"key":"17904_CR30","unstructured":"Sharma S (2019) Vgg16 and lstm image caption generator. https:\/\/www.kaggle.com\/code\/shweta2407\/vgg16-and-lstm-image-caption-generator\/notebook"},{"issue":"132","key":"17904_CR31","first-page":"306","volume":"404","author":"A Sherstinsky","year":"2020","unstructured":"Sherstinsky A (2020) Fundamentals of recurrent neural network (rnn) and long short-term memory (lstm) network. Physica D 404(132):306","journal-title":"Physica D"},{"key":"17904_CR32","unstructured":"Simonyan K, Zisserman A (2014) Very deep convolutional networks for large-scale image recognition. arXiv:1409.1556"},{"key":"17904_CR33","unstructured":"spaCy (2020) spacy. industrial-strength natural language processing in python. https:\/\/spacy.io\/"},{"key":"17904_CR34","unstructured":"Tang Y (2013) Deep learning using linear support vector machines. arXiv:1306.0239"},{"key":"17904_CR35","unstructured":"Vasiliev Y (2020) Natural language processing with Python and spaCy: A practical introduction. No Starch Press"},{"key":"17904_CR36","doi-asserted-by":"crossref","unstructured":"Vinyals O, Toshev A, Bengio S, et\u00a0al (2015) Show and tell: A neural image caption generator. In: Proceedings of the IEEE conference on computer vision and pattern recognition. pp 3156\u20133164","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"17904_CR37","doi-asserted-by":"crossref","unstructured":"Wang C, Yang H, Bartz C, et\u00a0al (2016) Image captioning with deep bidirectional lstms. In: Proceedings of the 24th ACM international conference on Multimedia. pp 988\u2013997","DOI":"10.1145\/2964284.2964299"},{"key":"17904_CR38","doi-asserted-by":"crossref","unstructured":"Wang H, Zhang Y, Yu X (2020) An overview of image caption generation methods. Comput Intel Neurosci 2020","DOI":"10.1155\/2020\/3062706"},{"issue":"102","key":"17904_CR39","first-page":"243","volume":"74","author":"H Wang","year":"2022","unstructured":"Wang H, Li Y, Xi S et al (2022) AMDCNet: An attentional multi-directional convolutional network for stereo matching. Displays 74(102):243","journal-title":"Displays"},{"key":"17904_CR40","unstructured":"Xiangyu (2022) Image caption using neural networks. https:\/\/xiangyutang2.github.io\/image-captioning\/"},{"key":"17904_CR41","unstructured":"Xu K, Ba J, Kiros R, et\u00a0al (2015) Show, attend and tell: Neural image caption generation with visual attention. In: International conference on machine learning. PMLR, pp 2048\u20132057"},{"key":"17904_CR42","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1162\/tacl_a_00166","volume":"2","author":"P Young","year":"2014","unstructured":"Young P, Lai A, Hodosh M et al (2014) From image descriptions to visual denotations: New similarity metrics for semantic inference over event descriptions. TACL 2:67\u201378","journal-title":"TACL"},{"key":"17904_CR43","doi-asserted-by":"crossref","unstructured":"Yu Z, Zhang C (2015) Image based static facial expression recognition with multiple deep network learning. In: Proceedings of the 2015 ACM on international conference on multimodal interaction. pp 435\u2013442","DOI":"10.1145\/2818346.2830595"},{"key":"17904_CR44","doi-asserted-by":"publisher","first-page":"103","DOI":"10.1016\/j.inffus.2020.01.011","volume":"59","author":"J Zhang","year":"2020","unstructured":"Zhang J, Yin Z, Chen P et al (2020) Emotion recognition using multi-modal data and machine learning techniques: A tutorial and review. Inform Fus 59:103\u2013126","journal-title":"Inform Fus"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-17904-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-023-17904-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-17904-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,3]],"date-time":"2024-09-03T02:11:40Z","timestamp":1725329500000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-023-17904-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,2,14]]},"references-count":44,"journal-issue":{"issue":"30","published-online":{"date-parts":[[2024,9]]}},"alternative-id":["17904"],"URL":"https:\/\/doi.org\/10.1007\/s11042-023-17904-3","relation":{},"ISSN":["1573-7721"],"issn-type":[{"type":"electronic","value":"1573-7721"}],"subject":[],"published":{"date-parts":[[2024,2,14]]},"assertion":[{"value":"22 June 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 December 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 December 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 February 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical Approval"}}]}}