{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,27]],"date-time":"2026-06-27T06:57:18Z","timestamp":1782543438692,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":35,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,5,13]],"date-time":"2024-05-13T00:00:00Z","timestamp":1715558400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,5,13]]},"DOI":"10.1145\/3589335.3651971","type":"proceedings-article","created":{"date-parts":[[2024,5,12]],"date-time":"2024-05-12T18:41:21Z","timestamp":1715539281000},"page":"1841-1848","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":28,"title":["Ensemble Pretrained Models for Multimodal Sentiment Analysis using Textual and Video Data Fusion"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-6894-0062","authenticated-orcid":false,"given":"Zhicheng","family":"Liu","sequence":"first","affiliation":[{"name":"School of Computer Science, The University of Sydney, Camperdown, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2561-6496","authenticated-orcid":false,"given":"Ali","family":"Braytee","sequence":"additional","affiliation":[{"name":"School of Computer Science, University of Technology, Sydney, Ultimo, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8864-0314","authenticated-orcid":false,"given":"Ali","family":"Anaissi","sequence":"additional","affiliation":[{"name":"School of Computer Science, The University of Sydney, Camperdown, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-6465-646X","authenticated-orcid":false,"given":"Guifu","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Computer Science, The University of Sydney, Camperdown, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-4009-4928","authenticated-orcid":false,"given":"Lingyun","family":"Qin","sequence":"additional","affiliation":[{"name":"School of Computer Science, The University of Sydney, Camperdown, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3719-338X","authenticated-orcid":false,"given":"Junaid","family":"Akram","sequence":"additional","affiliation":[{"name":"School of Computer Science, The University of Sydney, Camperdown, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,5,13]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1007\/s00530-010-0182-0"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"crossref","unstructured":"T. Baltru?aitis C. Ahuja and L. P. Morency. 2019. Multimodal Machine Learning: A Survey and Taxonomy. IEEE Transactions on Pattern Analysis and Machine Intelligence (2019).","DOI":"10.1109\/TPAMI.2018.2798607"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10579-008-9076-6"},{"key":"e_1_3_2_2_4_1","volume-title":"A transformer-based joint-encoding for emotion recognition and sentiment analysis. arXiv preprint arXiv:2006.15955","author":"Delbrouck Jean-Benoit","year":"2020","unstructured":"Jean-Benoit Delbrouck, No\u00e9 Tits, Mathilde Brousmiche, and St\u00e9phane Dupont. 2020. A transformer-based joint-encoding for emotion recognition and sentiment analysis. arXiv preprint arXiv:2006.15955 (2020)."},{"key":"e_1_3_2_2_5_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. arXiv preprint arXiv:1810.04805","author":"Devlin J.","year":"2018","unstructured":"J. Devlin, M. W. Chang, K. Lee, and K. Toutanova. 2018. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. arXiv preprint arXiv:1810.04805 (2018)."},{"key":"e_1_3_2_2_6_1","volume-title":"Ensemble Methods in Machine Learning. In International Workshop on Multiple Classifier Systems. Springer","author":"Dietterich T. G.","year":"2000","unstructured":"T. G. Dietterich. 2000. Ensemble Methods in Machine Learning. In International Workshop on Multiple Classifier Systems. Springer, Berlin, Heidelberg, 1--15."},{"key":"e_1_3_2_2_7_1","unstructured":"I. Goodfellow Y. Bengio and A. Courville. 2016. Deep Learning. MIT Press."},{"key":"e_1_3_2_2_8_1","volume-title":"Towards Arabic Multimodal Dataset for Sentiment Analysis. arXiv preprint arXiv:2306.06322","author":"Haouhat Abdelhamid","year":"2023","unstructured":"Abdelhamid Haouhat, Slimane Bellaouar, Attia Nehar, and Hadda Cherroun. 2023. Towards Arabic Multimodal Dataset for Sentiment Analysis. arXiv preprint arXiv:2306.06322 (2023)."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413678"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/T-AFFC.2011.15"},{"key":"e_1_3_2_2_11_1","first-page":"1097","article-title":"Imagenet Classification with Deep Convolutional Neural Networks","volume":"25","author":"Krizhevsky A.","year":"2012","unstructured":"A. Krizhevsky, I. Sutskever, and G. E. Hinton. 2012. Imagenet Classification with Deep Convolutional Neural Networks. Advances in Neural Information Processing Systems 25 (2012), 1097--1105.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1023\/A:1022859003006"},{"key":"e_1_3_2_2_13_1","first-page":"1449","article-title":"Multimodal Data Fusion","volume":"103","author":"Lahat D.","year":"2015","unstructured":"D. Lahat, T. Adali, and C. Jutten. 2015. Multimodal Data Fusion: An Overview of Methods, Challenges, and Prospects. Proc. IEEE 103, 9 (2015), 1449--1477.","journal-title":"An Overview of Methods, Challenges, and Prospects. Proc. IEEE"},{"key":"e_1_3_2_2_14_1","volume-title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations. arXiv preprint arXiv:1909.11942.","author":"Lan Z.","year":"2019","unstructured":"Z. Lan, M. Chen, S. Goodman, K. Gimpel, P. Sharma, and R. Soricut. 2019. ALBERT: A Lite BERT for Self-supervised Learning of Language Representations. arXiv preprint arXiv:1909.11942."},{"key":"e_1_3_2_2_15_1","unstructured":"Y. Liu M. Ott N. Goyal J. Du M. Joshi D. Chen and V. Stoyanov. 2019. RoBERTa: A Robustly Optimized BERT Pretraining Approach. arXiv preprint arXiv:1907.11692."},{"key":"e_1_3_2_2_16_1","volume-title":"Paul Pu Liang, Amir Zadeh, and Louis-Philippe Morency.","author":"Liu Zhun","year":"2018","unstructured":"Zhun Liu, Ying Shen, Varun Bharadhwaj Lakshminarasimhan, Paul Pu Liang, Amir Zadeh, and Louis-Philippe Morency. 2018. Efficient low-rank multimodal fusion with modality-specific factors. arXiv preprint arXiv:1806.00064 (2018)."},{"key":"e_1_3_2_2_17_1","first-page":"523","article-title":"Deep Convolutional Neural Networks Text-based Emotion Recognition","volume":"24","author":"Poria S.","year":"2017","unstructured":"S. Poria, E. Cambria, and A. Gelbukh. 2017. Deep Convolutional Neural Networks Text-based Emotion Recognition. IEEE Signal Processing Letters 24, 4 (2017), 523--527.","journal-title":"IEEE Signal Processing Letters"},{"key":"e_1_3_2_2_18_1","volume-title":"MELD: A Multimodal Multi-party Dataset for Emotion Recognition in Conversations. In ACL","author":"Poria S.","year":"2019","unstructured":"S. Poria, D. Hazarika, N. Majumder, G. Naik, E. Cambria, and R. Mihalcea. 2019. MELD: A Multimodal Multi-party Dataset for Emotion Recognition in Conversations. In ACL 2019."},{"key":"e_1_3_2_2_19_1","unstructured":"A. Radford L. Metz and S. Chintala. 2015. Unsupervised Representation Learning with Deep Convolutional Generative Adversarial Networks. arXiv preprint arXiv:1511.06434."},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.23919\/IConAC.2018.8748975"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1002\/widm.1249"},{"key":"e_1_3_2_2_22_1","unstructured":"V. Sanh L. Debut J. Chaumond and T.Wolf. 2019. DistilBERT a Distilled Version of BERT: Smaller Faster Cheaper and Lighter. arXiv preprint arXiv:1910.01108."},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6431"},{"key":"e_1_3_2_2_24_1","volume-title":"2018 International conference on frontiers of information technology (FIT). IEEE, 293--297","author":"Tahir Arsalan","year":"2018","unstructured":"Arsalan Tahir. 2018. Lexicon and heuristics based approach for identification of emotion in text. In 2018 International conference on frontiers of information technology (FIT). IEEE, 293--297."},{"key":"e_1_3_2_2_25_1","volume-title":"Proceedings of the conference. Association for Computational Linguistics. Meeting","volume":"2019","author":"Hubert Tsai Yao-Hung","year":"2019","unstructured":"Yao-Hung Hubert Tsai, Shaojie Bai, Paul Pu Liang, J Zico Kolter, Louis-Philippe Morency, and Ruslan Salakhutdinov. 2019. Multimodal transformer for unaligned multimodal language sequences. In Proceedings of the conference. Association for Computational Linguistics. Meeting, Vol. 2019. NIH Public Access, 6558."},{"key":"e_1_3_2_2_26_1","volume-title":"Advances in Neural Information Processing Systems","volume":"30","author":"Vaswani A.","year":"2017","unstructured":"A. Vaswani, N. Shazeer, N. Parmar, J. Uszkoreit, L. Jones, A. N. Gomez, et al. 2017. Attention is All You Need. In Advances in Neural Information Processing Systems, Vol. 30."},{"key":"e_1_3_2_2_27_1","volume-title":"On Deep Multi-view Representation Learning. In International Conference on Machine Learning. 1083--1092","author":"Wang W.","unstructured":"W. Wang, R. Arora, K. Livescu, and J. Bilmes. 2016. On Deep Multi-view Representation Learning. In International Conference on Machine Learning. 1083--1092."},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0893-6080(05)80023-1"},{"key":"e_1_3_2_2_29_1","volume-title":"Mtag: Modal-temporal attention graph for unaligned human multimodal language sequences. arXiv preprint arXiv:2010.11985","author":"Yang Jianing","year":"2020","unstructured":"Jianing Yang, Yongxin Wang, Ruitao Yi, Yuying Zhu, Azaan Rehman, Amir Zadeh, Soujanya Poria, and Louis-Philippe Morency. 2020. Mtag: Modal-temporal attention graph for unaligned human multimodal language sequences. arXiv preprint arXiv:2010.11985 (2020)."},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"crossref","unstructured":"A. Zadeh M. Chen S. Poria E. Cambria and L. P. Morency. 2016. Tensor Fusion Network for Multimodal Sentiment Analysis. arXiv preprint arXiv:1707.07250.","DOI":"10.18653\/v1\/D17-1115"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"crossref","unstructured":"A. Zadeh M. Chen S. Poria E. Cambria and L. P. Morency. 2017. Tensor Fusion Network for Multimodal Sentiment Analysis. In Emnlp. 1103--1114.","DOI":"10.18653\/v1\/D17-1115"},{"key":"e_1_3_2_2_32_1","volume-title":"Tensor fusion network for multimodal sentiment analysis. arXiv preprint arXiv:1707.07250","author":"Zadeh Amir","year":"2017","unstructured":"Amir Zadeh, Minghai Chen, Soujanya Poria, Erik Cambria, and Louis-Philippe Morency. 2017. Tensor fusion network for multimodal sentiment analysis. arXiv preprint arXiv:1707.07250 (2017)."},{"key":"e_1_3_2_2_33_1","volume-title":"Multimodal Language Analysis in the Wild: Carnegie Mellon University-MOSEI Dataset and Interpretable Dynamic Fusion Graph. In ACL","author":"Zadeh A.","year":"2018","unstructured":"A. Zadeh, P. P. Liang, S. Poria, E. Cambria, and L. P. Morency. 2018. Multimodal Language Analysis in the Wild: Carnegie Mellon University-MOSEI Dataset and Interpretable Dynamic Fusion Graph. In ACL 2018."},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/MIS.2016.94"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1208"}],"event":{"name":"WWW '24: The ACM Web Conference 2024","location":"Singapore Singapore","acronym":"WWW '24","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Companion Proceedings of the ACM Web Conference 2024"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3589335.3651971","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3589335.3651971","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:39:40Z","timestamp":1755823180000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3589335.3651971"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,13]]},"references-count":35,"alternative-id":["10.1145\/3589335.3651971","10.1145\/3589335"],"URL":"https:\/\/doi.org\/10.1145\/3589335.3651971","relation":{},"subject":[],"published":{"date-parts":[[2024,5,13]]},"assertion":[{"value":"2024-05-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}