{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,16]],"date-time":"2026-04-16T21:16:59Z","timestamp":1776374219712,"version":"3.51.2"},"publisher-location":"Singapore","reference-count":49,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819603503","type":"print"},{"value":"9789819603510","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,20]],"date-time":"2024-11-20T00:00:00Z","timestamp":1732060800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,20]],"date-time":"2024-11-20T00:00:00Z","timestamp":1732060800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-96-0351-0_21","type":"book-chapter","created":{"date-parts":[[2024,11,23]],"date-time":"2024-11-23T18:47:44Z","timestamp":1732387664000},"page":"281-297","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Multi-scale Cooperative Multimodal Transformers for\u00a0Multimodal Sentiment Analysis in\u00a0Videos"],"prefix":"10.1007","author":[{"given":"Lianyang","family":"Ma","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yu","family":"Yao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tao","family":"Liang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tongliang","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,20]]},"reference":[{"key":"21_CR1","doi-asserted-by":"crossref","unstructured":"Bagher\u00a0Zadeh, A., Liang, P.P., Poria, S., Cambria, E., Morency, L.P.: Multimodal language analysis in the wild: CMU-MOSEI dataset and interpretable dynamic fusion graph. In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 2236\u20132246 (2018)","DOI":"10.18653\/v1\/P18-1208"},{"key":"21_CR2","doi-asserted-by":"crossref","unstructured":"Busso, C., et al.: Iemocap: interactive emotional dyadic motion capture database. Lang. Resour. Eval. 42, 335\u2013359 (2008)","DOI":"10.1007\/s10579-008-9076-6"},{"key":"21_CR3","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"213","DOI":"10.1007\/978-3-030-58452-8_13","volume-title":"Computer Vision \u2013 ECCV 2020","author":"N Carion","year":"2020","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12346, pp. 213\u2013229. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58452-8_13"},{"key":"21_CR4","doi-asserted-by":"publisher","unstructured":"Chen, Y.-C., et al.: UNITER: UNiversal image-TExt representation learning. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12375, pp. 104\u2013120. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58577-8_7","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"21_CR5","doi-asserted-by":"crossref","unstructured":"Degottex, G., Kane, J., Drugman, T., Raitio, T., Scherer, S.: Covarep: a collaborative voice analysis repository for speech technologies (2014)","DOI":"10.1109\/ICASSP.2014.6853739"},{"key":"21_CR6","doi-asserted-by":"crossref","unstructured":"Delbrouck, J.B., Tits, N., Brousmiche, M., Dupont, S.: A transformer-based joint-encoding for emotion recognition and sentiment analysis. In: Second Grand-Challenge and Workshop on Multimodal Language (Challenge-HML), pp.\u00a01\u20137. Association for Computational Linguistics, Seattle (2020)","DOI":"10.18653\/v1\/2020.challengehml-1.1"},{"key":"21_CR7","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: BERT: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers), pp. 4171\u20134186. Association for Computational Linguistics (2019)"},{"key":"21_CR8","doi-asserted-by":"crossref","unstructured":"Dong, L., Xu, S., Xu, B.: Speech-transformer: a no-recurrence sequence-to-sequence model for speech recognition. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5884\u20135888 (2018)","DOI":"10.1109\/ICASSP.2018.8462506"},{"key":"21_CR9","unstructured":"Ekman, P.: Universal facial expressions of emotion. In: Culture and Personality: Contemporary Readings\/Chicago, pp. 12136\u201312145 (1974)"},{"key":"21_CR10","doi-asserted-by":"publisher","first-page":"424","DOI":"10.1016\/j.inffus.2022.09.025","volume":"91","author":"A Gandhi","year":"2023","unstructured":"Gandhi, A., Adhvaryu, K., Poria, S., Cambria, E., Hussain, A.: Multimodal sentiment analysis: a systematic review of history, datasets, multimodal fusion methods, applications, challenges and future directions. Information Fusion 91, 424\u2013444 (2023)","journal-title":"Information Fusion"},{"key":"21_CR11","doi-asserted-by":"crossref","unstructured":"Graves, A., Fern\u00e1ndez, S., Gomez, F., Schmidhuber, J.: Connectionist Temporal Classification: Labelling Unsegmented Sequence Data with Recurrent Neural Networks, vol.\u00a02006, pp. 369\u2013376 (2006)","DOI":"10.1145\/1143844.1143891"},{"key":"21_CR12","doi-asserted-by":"crossref","unstructured":"Hazarika, D., Zimmermann, R., Poria, S.: Misa: modality-invariant and-specific representations for multimodal sentiment analysis. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 1122\u20131131 (2020)","DOI":"10.1145\/3394171.3413678"},{"key":"21_CR13","unstructured":"Hou, M., Tang, J., Zhang, J., Kong, W., Zhao, Q.: Deep multimodal multilinear fusion with high-order polynomial pooling. In: Advances in Neural Information Processing Systems, vol.\u00a032, pp. 12136\u201312145 (2019)"},{"key":"21_CR14","doi-asserted-by":"crossref","unstructured":"Hu, R., Singh, A., Darrell, T., Rohrbach, M.: Iterative answer prediction with pointer-augmented multimodal transformers for textvqa. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 9989\u20139999 (2020)","DOI":"10.1109\/CVPR42600.2020.01001"},{"key":"21_CR15","unstructured":"iMotions. Facial expression analysis (2017). https:\/\/imotions.com\/biosensor\/fea-facial-expression-analysis\/"},{"key":"21_CR16","doi-asserted-by":"crossref","unstructured":"Liang, P.P., Liu, Z., Bagher\u00a0Zadeh, A., Morency, L.P.: Multimodal language analysis with recurrent multistage fusion. In: Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing, pp. 150\u2013161 (2018)","DOI":"10.18653\/v1\/D18-1014"},{"key":"21_CR17","unstructured":"Lu, J., Batra, D., Parikh, D., Lee, S.: Vilbert: pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. In: Advances in Neural Information Processing Systems, vol.\u00a032, pp. 13\u201323 (2019)"},{"key":"21_CR18","doi-asserted-by":"crossref","unstructured":"Luo, H., et al.: Clip4clip: an empirical study of clip for end to end video clip retrieval and captioning. Neurocomputing 508, 293\u2013304 (2022)","DOI":"10.1016\/j.neucom.2022.07.028"},{"key":"21_CR19","doi-asserted-by":"crossref","unstructured":"Ma, J., Bai, Y., Zhong, B., Zhang, W., Yao, T., Mei, T.: Visualizing and understanding patch interactions in vision transformer. IEEE Transactions on Neural Networks and Learning Systems (2023)","DOI":"10.1109\/TNNLS.2023.3270479"},{"issue":"9","key":"21_CR20","doi-asserted-by":"publisher","first-page":"11297","DOI":"10.1109\/TPAMI.2023.3266023","volume":"45","author":"J Miao","year":"2023","unstructured":"Miao, J., Wei, Y., Wang, X., Yang, Y.: Temporal pixel-level semantic understanding through the vspw dataset. IEEE Trans. Pattern Anal. Mach. Intell. 45(9), 11297\u201311308 (2023)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"21_CR21","doi-asserted-by":"crossref","unstructured":"Pang, B., Lee, L.: A sentimental education: sentiment analysis using subjectivity summarization based on minimum cuts. In: Proceedings of the 42nd Annual Meeting on Association for Computational Linguistics (ACL 2004). Association for Computational Linguistics (2004)","DOI":"10.3115\/1218955.1218990"},{"key":"21_CR22","unstructured":"Parmar, N., et al.: Image transformer. In: Dy, J., Krause, A. (eds.) Proceedings of Machine Learning Research, vol.\u00a080, pp. 4055\u20134064. PMLR (2018)"},{"key":"21_CR23","doi-asserted-by":"crossref","unstructured":"Pennington, J., Socher, R., Manning, C.D.: Glove: global vectors for word representation. In: Empirical Methods in Natural Language Processing (EMNLP), pp. 1532\u20131543 (2014)","DOI":"10.3115\/v1\/D14-1162"},{"key":"21_CR24","doi-asserted-by":"crossref","unstructured":"Pham, H., Liang, P., Manzini, T., Morency, L.P., Poczos, B.: Found in translation: Learning robust joint representations by cyclic translations between modalities. Proc. AAAI Conf. Artif. Intell. 33, 6892\u20136899 (2019)","DOI":"10.1609\/aaai.v33i01.33016892"},{"key":"21_CR25","doi-asserted-by":"crossref","unstructured":"Pham, H., Manzini, T., Liang, P.P., Pocz\u00f3s, B.: Seq2Seq2Sentiment: multimodal sequence to sequence models for sentiment analysis. In: Proceedings of Grand Challenge and Workshop on Human Multimodal Language (Challenge-HML), pp. 53\u201363. Association for Computational Linguistics, Melbourne (2018)","DOI":"10.18653\/v1\/W18-3308"},{"key":"21_CR26","doi-asserted-by":"crossref","unstructured":"Poria, S., Cambria, E., Hazarika, D., Majumder, N., Zadeh, A., Morency, L.P.: Context-dependent sentiment analysis in user-generated videos. In: Proceedings of the 55th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 873\u2013883 (2017)","DOI":"10.18653\/v1\/P17-1081"},{"key":"21_CR27","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"21_CR28","doi-asserted-by":"crossref","unstructured":"Rahman, T., Busso, C.: A personalized emotion recognition system using an unsupervised feature adaptation scheme. In: 2012 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5117\u20135120 (2012)","DOI":"10.1109\/ICASSP.2012.6289072"},{"key":"21_CR29","doi-asserted-by":"crossref","unstructured":"Rahman, W., et al.: Integrating multimodal information in large pretrained transformers. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, pp. 2359\u20132369 (2020)","DOI":"10.18653\/v1\/2020.acl-main.214"},{"key":"21_CR30","doi-asserted-by":"crossref","unstructured":"Shenoy, A., Sardana, A.: Multilogue-net: a context-aware RNN for multi-modal emotion detection and sentiment analysis in conversation. In: Second Grand-Challenge and Workshop on Multimodal Language (Challenge-HML). Association for Computational Linguistics, Seattle (2020)","DOI":"10.18653\/v1\/2020.challengehml-1.3"},{"key":"21_CR31","doi-asserted-by":"crossref","unstructured":"Shenoy, A., Sardana, A., Graphics, N.: Multilogue-net: A Context Aware RNN for Multi-modal Emotion Detection and Sentiment Analysis in Conversation, p.\u00a019 (2020)","DOI":"10.18653\/v1\/2020.challengehml-1.3"},{"issue":"9","key":"21_CR32","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3652149","volume":"56","author":"U Singh","year":"2024","unstructured":"Singh, U., Abhishek, K., Azad, H.K.: A survey of cutting-edge multimodal sentiment analysis. ACM Comput. Surv. 56(9), 1\u201338 (2024)","journal-title":"ACM Comput. Surv."},{"key":"21_CR33","doi-asserted-by":"crossref","unstructured":"Socher, R., et al.: Recursive deep models for semantic compositionality over a sentiment treebank. In: Proceedings of the 2013 Conference on Empirical Methods in Natural Language Processing (2013)","DOI":"10.18653\/v1\/D13-1170"},{"key":"21_CR34","doi-asserted-by":"crossref","unstructured":"Tan, H., Bansal, M.: LXMERT: learning cross-modality encoder representations from transformers. In: Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP), pp. 5100\u20135111. Association for Computational Linguistics (2019)","DOI":"10.18653\/v1\/D19-1514"},{"key":"21_CR35","doi-asserted-by":"crossref","unstructured":"Tsai, Y.H.H., Bai, S., Liang, P.P., Kolter, J.Z., Morency, L.P., Salakhutdinov, R.: Multimodal transformer for unaligned multimodal language sequences. In: Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). Association for Computational Linguistics (2019)","DOI":"10.18653\/v1\/P19-1656"},{"key":"21_CR36","unstructured":"Tsai, Y.H.H., Liang, P.P., Zadeh, A., Morency, L.P., Salakhutdinov, R.: Learning factorized multimodal representations. In: ICLR (2019)"},{"key":"21_CR37","doi-asserted-by":"crossref","unstructured":"Tsai, Y.H.H., Ma, M., Yang, M., Salakhutdinov, R., Morency, L.P.: Multimodal routing: improving local and global interpretability of multimodal language analysis. In: Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP), pp. 1823\u20131833 (2020)","DOI":"10.18653\/v1\/2020.emnlp-main.143"},{"key":"21_CR38","doi-asserted-by":"crossref","unstructured":"Turney, P.D.: Thumbs up or thumbs down? semantic orientation applied to unsupervised classification of reviews. In: Proceedings of the 40th Annual Meeting on Association for Computational Linguistics. Association for Computational Linguistics (2002)","DOI":"10.3115\/1073083.1073153"},{"key":"21_CR39","unstructured":"Vaswani, A., et al.: Attention is all you need. Adv. Neural Inf. Process. Syst. 30, 5998\u20136008 (2017)"},{"key":"21_CR40","doi-asserted-by":"crossref","unstructured":"Wang, H., Meghawat, A., Morency, L., Xing, E.P.: Select-additive learning: improving generalization in multimodal sentiment analysis. In: 2017 IEEE International Conference on Multimedia and Expo (ICME) (2017)","DOI":"10.1109\/ICME.2017.8019301"},{"key":"21_CR41","doi-asserted-by":"crossref","unstructured":"Wang, Y., Shen, Y., Liu, Z., Liang, P., Zadeh, A., Morency, L.P.: Words can shift: dynamically adjusting word representations using nonverbal behaviors. Proc. AAAI Conf. Artif. Intell. 33, 7216\u20137223 (2019)","DOI":"10.1609\/aaai.v33i01.33017216"},{"key":"21_CR42","doi-asserted-by":"crossref","unstructured":"Wang, Y., et\u00a0al.: Internvideo2: scaling video foundation models for multimodal video understanding. arXiv preprint arXiv:2403.15377 (2024)","DOI":"10.1007\/978-3-031-73013-9_23"},{"key":"21_CR43","doi-asserted-by":"crossref","unstructured":"Yenduri, G., et\u00a0al.: GPT (generative pre-trained transformer)\u2013a comprehensive review on enabling technologies, potential applications, emerging challenges, and future directions. IEEE Access (2024)","DOI":"10.1109\/ACCESS.2024.3389497"},{"key":"21_CR44","doi-asserted-by":"crossref","unstructured":"Yu, W., Xu, H., Yuan, Z., Wu, J.: Learning modality-specific representations with self-supervised multi-task learning for multimodal sentiment analysis. Proc. AAAI Conf. Artif. Intell. 35, 10790\u201310797 (2021)","DOI":"10.1609\/aaai.v35i12.17289"},{"key":"21_CR45","doi-asserted-by":"crossref","unstructured":"Yuan, J., Liberman, M.: Speaker identification on the scotus corpus. J. Acoust. Soc. Am. 123, 3878 (2008)","DOI":"10.1121\/1.2935783"},{"issue":"6","key":"21_CR46","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1109\/MIS.2016.94","volume":"31","author":"A Zadeh","year":"2016","unstructured":"Zadeh, A., Zellers, R., Pincus, E., Morency, L.: Multimodal sentiment intensity analysis in videos: facial gestures and verbal messages. IEEE Intell. Syst. 31(6), 82\u201388 (2016)","journal-title":"IEEE Intell. Syst."},{"key":"21_CR47","doi-asserted-by":"crossref","unstructured":"Zadeh, A., Chen, M., Poria, S., Cambria, E., Morency, L.P.: Tensor fusion network for multimodal sentiment analysis. In: Proceedings of the 2017 Conference on Empirical Methods in Natural Language Processing, pp. 1103\u20131114 (2017)","DOI":"10.18653\/v1\/D17-1115"},{"key":"21_CR48","unstructured":"Zhu, J., et al.: Vl-gpt: a generative pre-trained transformer for vision and language understanding and generation. arXiv preprint arXiv:2312.09251 (2023)"},{"key":"21_CR49","doi-asserted-by":"publisher","first-page":"3375","DOI":"10.1109\/TMM.2022.3160060","volume":"25","author":"T Zhu","year":"2022","unstructured":"Zhu, T., Li, L., Yang, J., Zhao, S., Liu, H., Qian, J.: Multimodal sentiment analysis with image-text interaction network. IEEE Trans. Multimedia 25, 3375\u20133385 (2022)","journal-title":"IEEE Trans. Multimedia"}],"container-title":["Lecture Notes in Computer Science","AI 2024: Advances in Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-96-0351-0_21","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T22:07:41Z","timestamp":1733090861000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-96-0351-0_21"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,20]]},"ISBN":["9789819603503","9789819603510"],"references-count":49,"URL":"https:\/\/doi.org\/10.1007\/978-981-96-0351-0_21","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,20]]},"assertion":[{"value":"20 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"AI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Australasian Joint Conference on Artificial Intelligence","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Melbourne, VIC","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Australia","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25 November 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 November 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"37","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ausai2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ajcai2024.org\/index.html","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}