{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T22:59:29Z","timestamp":1783983569632,"version":"3.55.0"},"reference-count":54,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2023,9,5]],"date-time":"2023-09-05T00:00:00Z","timestamp":1693872000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,9,5]],"date-time":"2023-09-05T00:00:00Z","timestamp":1693872000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100004541","name":"Ministry of Education, India","doi-asserted-by":"publisher","award":["1-3146198040"],"award-info":[{"award-number":["1-3146198040"]}],"id":[{"id":"10.13039\/501100004541","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-023-16443-1","type":"journal-article","created":{"date-parts":[[2023,9,5]],"date-time":"2023-09-05T10:02:48Z","timestamp":1693908168000},"page":"28373-28394","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":44,"title":["Interpretable multimodal emotion recognition using hybrid fusion of speech and image data"],"prefix":"10.1007","volume":"83","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4318-1353","authenticated-orcid":false,"given":"Puneet","family":"Kumar","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sarthak","family":"Malik","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Balasubramanian","family":"Raman","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,9,5]]},"reference":[{"key":"16443_CR1","doi-asserted-by":"crossref","unstructured":"Baltru\u0161aitis T, Ahuja C, Morency L-P (2018) Multimodal machine learning: a survey and taxonomy. IEEE Trans Pattern Anal Mach Intell (T-PAMI) 41(2):423\u2013443","DOI":"10.1109\/TPAMI.2018.2798607"},{"key":"16443_CR2","doi-asserted-by":"crossref","unstructured":"Busso C, Bulut M, Lee C-C, Kazemzadeh A, Mower E, Kim S, Chang J-N, Lee S, Narayanan S-S (2008) IEMOCAP: Interactive Emotional dyadic MOtion CAPture data. Lang Resour Eval 42(4)","DOI":"10.1007\/s10579-008-9076-6"},{"key":"16443_CR3","doi-asserted-by":"crossref","unstructured":"Chan W, Jaitly N, Le Q, Vinyals O (2016) Listen, attend and spell: a neural network for large vocabulary conversational speech recognition. In IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). pp 4960\u20134964","DOI":"10.1109\/ICASSP.2016.7472621"},{"key":"16443_CR4","doi-asserted-by":"crossref","unstructured":"Dai D, Wu Z, Li R, Wu X, Jia J, Meng H (2019) Learning discriminative features from spectrograms using center loss for SER. In IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). pp 7405\u20137409","DOI":"10.1109\/ICASSP.2019.8683765"},{"key":"16443_CR5","unstructured":"Deep Mind (2016) Wavenet: a generative model for raw audio. http:\/\/deepmind.com\/blog\/article\/wavenet-generative-model-raw-audio. Accessed on 20 Feb 2022"},{"key":"16443_CR6","doi-asserted-by":"crossref","unstructured":"Fan S, Lin C, Li H, Lin Z, Su J, Zhang H, Gong Y, Guo J, Duan N (2022) Sentiment aware word and sentence level pre-training for sentiment analysis. In Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing (EMNLP). pp 4984\u20134994","DOI":"10.18653\/v1\/2022.emnlp-main.332"},{"key":"16443_CR7","doi-asserted-by":"crossref","unstructured":"Finka L-R, Luna S-P, Brondani J-T, Tzimiropoulos Y, McDonagh J, Farnworth M-J, Ruta M, Mills D-S (2019) Geometric morphometrics for the study of facial expressions in non-human animals, using the domestic cat as an exemplar. Sci Rep 9(1):9883","DOI":"10.1038\/s41598-019-46330-5"},{"key":"16443_CR8","doi-asserted-by":"crossref","unstructured":"Gaspar A, Alexandre L-A (2019) A multimodal approach to image sentiment analysis. In Springer International Conference on Intelligent Data Engineering and Automated Learning (IDEAL). pp 302\u2013309","DOI":"10.1007\/978-3-030-33607-3_33"},{"key":"16443_CR9","doi-asserted-by":"crossref","unstructured":"Guanghui C, Xiaoping Z (2021) Multimodal emotion recognition by fusing correlation features of speech-visual. IEEE Signal Process Lett 28:533\u2013537","DOI":"10.1109\/LSP.2021.3055755"},{"key":"16443_CR10","doi-asserted-by":"crossref","unstructured":"Han W, Chen H, Poria S (2021) Improving multimodal fusion with hierarchical mutual information maximization for multimodal sentiment analysis. In Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing (EMNLP). pp 9180\u20139192","DOI":"10.18653\/v1\/2021.emnlp-main.723"},{"key":"16443_CR11","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep Residual Learning for Image Recognition. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"16443_CR12","doi-asserted-by":"crossref","unstructured":"Hossain M-S, Muhammad G (2019) Emotion recognition using deep learning approach from audio visual emotional big data. Inf Fusion 49:69\u201378","DOI":"10.1016\/j.inffus.2018.09.008"},{"key":"16443_CR13","unstructured":"Howard A-G, Zhu M, Chen B, Kalenichenko D, Wang W, Weyand T, Andreetto M, Adam H (2017) MobileNets: efficient convolutional neural networks for mobile vision applications. arXiv:1704.04861. Accessed 06 Jan 2023"},{"key":"16443_CR14","doi-asserted-by":"crossref","unstructured":"Huang G, Liu Z, Van Der\u00a0Maaten L, Weinberger K-Q (2017) Densely Connected Convolutional Networks. In Proceedings of the IEEE\/CVF conference on Computer Vision and Pattern Recognition (CVPR). pp 4700\u20134708","DOI":"10.1109\/CVPR.2017.243"},{"key":"16443_CR15","doi-asserted-by":"crossref","unstructured":"Hu A, Flaxman S (2018) Multimodal sentiment analysis to explore the structure of emotions. In ACM SIGKDD International Conference on Knowledge Discovery & Data Mining (KDD). pp 350\u2013358","DOI":"10.1145\/3219819.3219853"},{"key":"16443_CR16","doi-asserted-by":"crossref","unstructured":"Kim H-R, Kim Y-S, Kim S-J, Lee I-K (2018) Building emotional machines: recognizing image emotions through deep neural networks. IEEE Trans Multimed (T-MM) 20(11):2980\u20132992","DOI":"10.1109\/TMM.2018.2827782"},{"key":"16443_CR17","doi-asserted-by":"crossref","unstructured":"Kumar P, Jain S, Raman B, Roy P-P, Iwamura M (2021) End-to-end triplet loss based emotion embedding system for speech emotion recognition. In IEEE International Conference on Pattern Recognition (ICPR). pp 8766\u20138773","DOI":"10.1109\/ICPR48806.2021.9413144"},{"key":"16443_CR18","doi-asserted-by":"crossref","unstructured":"Kumar P, Kaushik V, Raman B (2021) Towards the explainability of multimodal speech emotion recognition. In INTERSPEECH. pp 1748\u20131752","DOI":"10.21437\/Interspeech.2021-1718"},{"key":"16443_CR19","doi-asserted-by":"crossref","unstructured":"Kumar P, Khokher V, Gupta Y, Raman B (2021) Hybrid fusion based approach for multimodal emotion recognition with insufficient labeled data. In 2021 IEEE International Conference on Image Processing (ICIP). IEEE, pp 314\u2013318","DOI":"10.1109\/ICIP42928.2021.9506714"},{"key":"16443_CR20","doi-asserted-by":"crossref","unstructured":"Kwon S (2019) A CNN assisted enhanced audio signal processing for speech emotion recognition. Sensors 20(1):183","DOI":"10.3390\/s20010183"},{"key":"16443_CR21","doi-asserted-by":"crossref","unstructured":"Lin T-Y, Goyal P, Girshick R, He K, Doll\u00e1r P (2017) Focal loss for dense object detection. In Proceedings of the IEEE\/CVF Conference on Computer Vision (ICCV). pp 2980\u20132988","DOI":"10.1109\/ICCV.2017.324"},{"key":"16443_CR22","doi-asserted-by":"crossref","unstructured":"Lu X, Adams R-B, Li J, Newman M-G, Wang J-Z (2017) An investigation into three visual characteristics of complex scenes that evoke human emotion. In The Seventh International Conference on Affective Computing and Intelligent Interaction (ACII). IEEE, pp 440\u2013447","DOI":"10.1109\/ACII.2017.8273637"},{"key":"16443_CR23","unstructured":"Lundberg S-M, Lee S-I (2017) A unified approach to interpreting model predictions. In The 31st International Conference on Neural Information Processing Systems (NeurIPS). pp 4768\u20134777"},{"key":"16443_CR24","doi-asserted-by":"crossref","unstructured":"Lu X, Wang W, Ma C, Shen J, Shao L, Porikli F (2019) See more, know more: unsupervised video object segmentation with co-attention siamese networks. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). pp 3623\u20133632","DOI":"10.1109\/CVPR.2019.00374"},{"key":"16443_CR25","doi-asserted-by":"crossref","unstructured":"Lu X, Wang W, Shen J, Crandall D-J, Gool L-V (2021) Segmenting objects from relational visual data. IEEE Trans Pattern Anal Mach Intell (T-PAMI) 44(11):7885\u20137897","DOI":"10.1109\/TPAMI.2021.3115815"},{"key":"16443_CR26","doi-asserted-by":"crossref","unstructured":"Lu X, Wang W, Shen J, Crandall D, Luo J (2020) Zero shot video object segmentation with co-attention siamese networks. IEEE Trans Pattern Anal Mach Intell (T-PAMI) 44(4):2228\u20132242","DOI":"10.1109\/TPAMI.2020.3040258"},{"key":"16443_CR27","doi-asserted-by":"crossref","unstructured":"Maji B, Swain M (2022) Advanced fusion-based speech emotion recognition system using a dual attention mechanism with conv-caps and Bi-GRU features. Electron 11(9):1328","DOI":"10.3390\/electronics11091328"},{"key":"16443_CR28","doi-asserted-by":"crossref","unstructured":"Majumder N, Poria S, Hazarika D, Mihalcea R, Gelbukh A, Cambria E (2019) DialogueRNN: an attentive RNN for emotion detection in conversations. In Conference on Artificial Intelligence (AAAI) 33:6818\u20136825","DOI":"10.1609\/aaai.v33i01.33016818"},{"key":"16443_CR29","doi-asserted-by":"crossref","unstructured":"Makiuchi M-R, Uto K, Shinoda K (2021) Multimodal emotion recognition with high-level speech and text features. In IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)","DOI":"10.1109\/ASRU51503.2021.9688036"},{"key":"16443_CR30","doi-asserted-by":"crossref","unstructured":"Malik S, Kumar P, Raman B (2021) Towards interpretable facial emotion recognition. In The 12th Indian Conference on Computer Vision, Graphics and Image Processing (ICVGIP). pp 1\u20139","DOI":"10.1145\/3490035.3490271"},{"key":"16443_CR31","unstructured":"Opitz J, Burst S (2019) Macro f1 and Macro f1. arXiv:1911.03347"},{"key":"16443_CR32","doi-asserted-by":"crossref","unstructured":"Pag\u00e9\u00a0Fortin M, Chaib-draa B (2019) Multimodal multitask emotion recognition using images, texts and tags. In ACM workshop on cross-modal learning and application. pp 3\u201310","DOI":"10.1145\/3326459.3329165"},{"key":"16443_CR33","unstructured":"Ping W, Peng K, Gibiansky A, Arik S-O, Kannan A, Narang S, Raiman J, Miller J (2018) DeepVoice 3: scaling text-to-speech with convolutional sequence learning. In The 6th Int. Conference on Learning Representations (ICLR)"},{"key":"16443_CR34","doi-asserted-by":"crossref","unstructured":"Plutchik R (2001) The nature of emotions. J Stor Digit Lib Am Sci J 89(4):344\u2013350","DOI":"10.1511\/2001.4.344"},{"key":"16443_CR35","doi-asserted-by":"crossref","unstructured":"Poria S, Cambria E, Bajpai R, Hussain A (2017) A review of affective computing: from unimodal analysis to multimodal fusion. Elsevier Inf Fus J 37:98\u2013125","DOI":"10.1016\/j.inffus.2017.02.003"},{"key":"16443_CR36","doi-asserted-by":"crossref","unstructured":"Rao T, Li X, Xu M (2019) Learning multi-level deep representations for image emotion classification. Neural Process Lett 1\u201319","DOI":"10.1007\/s11063-019-10033-9"},{"key":"16443_CR37","doi-asserted-by":"crossref","unstructured":"Ribeiro M-T, Singh S, Guestrin C (2016) Why should i trust you? Explaining predictions of any classifier. In International Conference on Knowledge Discovery & Data mining (KDD). pp 1135\u20131144","DOI":"10.1145\/2939672.2939778"},{"key":"16443_CR38","doi-asserted-by":"crossref","unstructured":"Salamon J, Bello J-P (2017) Deep convolutional neural networks and data augmentation for environmental sound classification. IEEE Signal Process Lett 24(3):279\u2013283","DOI":"10.1109\/LSP.2017.2657381"},{"key":"16443_CR39","doi-asserted-by":"crossref","unstructured":"Selvaraju R-R, Cogswell M, Das A, Vedantam R, Parikh D, Batra D (2017) Grad-CAM: visual explanations from deep networks via gradient-based localization. In The IEEE\/CVF International Conference on Computer Vision (ICCV). pp 618\u2013626","DOI":"10.1109\/ICCV.2017.74"},{"key":"16443_CR40","unstructured":"Shrikumar A, Greenside P, Kundaje A (2017) Learning Important Features Through Propagating Activation Differences. In International Conference on Machine Learning (ICML). pp 3145\u20133153"},{"key":"16443_CR41","unstructured":"Simonyan K, Zisserman A (2014) Very deep convolutional networks for large-scale image recognition. arXiv:1409.1556. Accessed 06 Jan 2023"},{"key":"16443_CR42","doi-asserted-by":"crossref","unstructured":"Siriwardhana S, Reis A, Weerasekera R (2020) Jointly fine tuning \u2018BERT-Like\u2019 Self supervised models to improve multimodal speech emotion recognition. INTERSPEECH pp 3755\u20133759","DOI":"10.21437\/Interspeech.2020-1212"},{"key":"16443_CR43","doi-asserted-by":"crossref","unstructured":"Szegedy C, Liu W, Jia Y, Sermanet P, Reed S, Anguelov D, Erhan D, Vanhoucke V, Rabinovich A (2015) Going deeper with convolutions. In Proceedings of the IEEE\/CVF conference on Computer Vision and Pattern Recognition (CVPR). pp 1\u20139","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"16443_CR44","unstructured":"Tan M, Le Q (2019) EfficientNet: rethinking model scaling for CNN. In International Conference on Machine Learning (ICML). pp 6105\u20136114"},{"key":"16443_CR45","doi-asserted-by":"crossref","unstructured":"Teng J, Lu X, Gong Y, Liu X, Nie X, Yin Y (2021) Regularized two granularity loss function for weakly supervised video moment retrieval. IEEE Trans Multimed (T-MM) 24:1141\u20131151","DOI":"10.1109\/TMM.2021.3120545"},{"key":"16443_CR46","doi-asserted-by":"crossref","unstructured":"Vadicamo L, Carrara F, Cimino A, Cresci S, Dell\u2019Orletta F, Falchi F, Tesconi M (2017) Cross-media learning for image sentiment analysis in the wild. In IEEE International Conference on Computer Vision Workshops (ICCV-W). pp 308\u2013317","DOI":"10.1109\/ICCVW.2017.45"},{"key":"16443_CR47","unstructured":"van\u00a0den Oord A, Dieleman S, Zen H, Simonyan K, Vinyals O, Graves A, Kalchbrenner N, Senior A, Kavukcuoglu K (2016) Wavenet: a generative model for raw audio. arXiv:1609.03499"},{"key":"16443_CR48","doi-asserted-by":"crossref","unstructured":"Vieira S-M, Kaymak U, Sousa J-MC (2010) Cohen\u2019s kappa coefficient as a performance measure for feature selection. In International Conference on Fuzzy Systems. IEEE, pp 1\u20138","DOI":"10.1109\/FUZZY.2010.5584447"},{"key":"16443_CR49","doi-asserted-by":"crossref","unstructured":"Xu M, Zhang F, Khan S-U (2020) Improve accuracy of speech emotion recognition with attention head fusion. In IEEE Annual Computing and Communication Workshop and Conference (CCWC). pp 1058\u20131064","DOI":"10.1109\/CCWC47524.2020.9031207"},{"key":"16443_CR50","doi-asserted-by":"crossref","unstructured":"Yenigalla P, Kumar A, Tripathi S, Singh C, Kar S, Vepa J (2018) Speech Emotion Recognition Using Spectrogram & Phoneme Embedding. In INTERSPEECH pp 3688\u20133692","DOI":"10.21437\/Interspeech.2018-1811"},{"key":"16443_CR51","doi-asserted-by":"crossref","unstructured":"You Q, Luo J, Jin H, Yang J (2016) Building a large scale dataset for image emotion recognition: the fine print and the benchmark. In The 30th AAAI Conference on Artificial Intelligence (AAAI). pp 308\u2013314","DOI":"10.1609\/aaai.v30i1.9987"},{"key":"16443_CR52","doi-asserted-by":"crossref","unstructured":"Zeng Y, Li Z, Chen Z, Ma H (2023) Aspect-level sentiment analysis based on semantic heterogeneous graph convolutional network. Front Comput Sci 17(6):176340","DOI":"10.1007\/s11704-022-2256-5"},{"key":"16443_CR53","doi-asserted-by":"crossref","unstructured":"Zeng Y, Li Z, Tang Z, Chen Z, Ma H (2023) Heterogeneous graph convolution based on in-domain self-supervision for multimodal sentiment analysis. Exp Syst Appl 213:119240","DOI":"10.1016\/j.eswa.2022.119240"},{"key":"16443_CR54","doi-asserted-by":"crossref","unstructured":"Zeng Z, Pantic M, Roisman G-I, Huang T-S (2009) A survey of affect recognition: audio, visual, and spontaneous expressions. IEEE Trans Pattern Anal Mach Intell (T-PAMI) 31(1):39\u201358","DOI":"10.1109\/TPAMI.2008.52"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-16443-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-023-16443-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-16443-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,7]],"date-time":"2024-03-07T08:09:54Z","timestamp":1709798994000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-023-16443-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,9,5]]},"references-count":54,"journal-issue":{"issue":"10","published-online":{"date-parts":[[2024,3]]}},"alternative-id":["16443"],"URL":"https:\/\/doi.org\/10.1007\/s11042-023-16443-1","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,9,5]]},"assertion":[{"value":"11 January 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 July 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 August 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 September 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interest"}},{"value":"Authors have no conflict of interest.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of interest"}},{"value":"Not applicable","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval"}},{"value":"Not applicable","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}},{"value":"Not applicable","order":6,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"This article does not contain any studies with human participants or animals performed by any of the authors.","order":7,"name":"Ethics","group":{"name":"EthicsHeading","label":"Informed consent"}}]}}