{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T20:26:05Z","timestamp":1782937565806,"version":"3.54.5"},"publisher-location":"Singapore","reference-count":36,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819785018","type":"print"},{"value":"9789819785025","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-97-8502-5_30","type":"book-chapter","created":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T14:03:04Z","timestamp":1730383384000},"page":"423-437","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["CFMISA: Cross-Modal Fusion of\u00a0Modal Invariant and\u00a0Specific Representations for\u00a0Multimodal Sentiment Analysis"],"prefix":"10.1007","author":[{"given":"Haiying","family":"Xia","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jingwen","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yumei","family":"Tan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaohu","family":"Tang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,11,1]]},"reference":[{"issue":"9","key":"30_CR1","doi-asserted-by":"publisher","first-page":"28077","DOI":"10.1007\/s11042-023-16560-x","volume":"83","author":"L Agarwal","year":"2024","unstructured":"Agarwal, L., Verma, B.: From methods to datasets: a survey on image-caption generators. Multimedia Tools Appl 83(9), 28077\u201328123 (2024)","journal-title":"Multimedia Tools Appl"},{"key":"30_CR2","doi-asserted-by":"crossref","unstructured":"Brave, S., Nass, C.: Emotion in human-computer interaction. In: The Human-Computer Interaction Handbook, pp. 103\u2013118. CRC Press (2007)","DOI":"10.1201\/9781410615862-13"},{"key":"30_CR3","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: pre-training of deep bidirectional transformers for language understanding (2018). arXiv preprint arXiv:1810.04805"},{"key":"30_CR4","doi-asserted-by":"crossref","unstructured":"Doctor, F., Karyotis, C., Iqbal, R., James, A.: An intelligent framework for emotion aware e-healthcare support systems. In: 2016 IEEE Symposium Series on Computational Intelligence (SSCI), pp.\u00a01\u20138. IEEE (2016)","DOI":"10.1109\/SSCI.2016.7850044"},{"key":"30_CR5","doi-asserted-by":"crossref","unstructured":"Girdhar, R., El-Nouby, A., Liu, Z., Singh, M., Alwala, K.V., Joulin, A., Misra, I.: Imagebind: one embedding space to bind them all. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15180\u201315190 (2023)","DOI":"10.1109\/CVPR52729.2023.01457"},{"key":"30_CR6","doi-asserted-by":"crossref","unstructured":"Hazarika, D., Zimmermann, R., Poria, S.: Misa: modality-invariant and-specific representations for multimodal sentiment analysis. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 1122\u20131131 (2020)","DOI":"10.1145\/3394171.3413678"},{"issue":"8","key":"30_CR7","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., Schmidhuber, J.: Long short-term memory. Neural Comput. 9(8), 1735\u20131780 (1997)","journal-title":"Neural Comput."},{"key":"30_CR8","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2023.126992","volume":"565","author":"J Huang","year":"2024","unstructured":"Huang, J., Pu, Y., Zhou, D., Cao, J., Gu, J., Zhao, Z., Xu, D.: Dynamic hypergraph convolutional network for multimodal sentiment analysis. Neurocomputing 565, 126992 (2024)","journal-title":"Neurocomputing"},{"key":"30_CR9","first-page":"30291","volume":"35","author":"P Jin","year":"2022","unstructured":"Jin, P., Huang, J., Liu, F., Wu, X., Ge, S., Song, G., Clifton, D., Chen, J.: Expectation-maximization contrastive learning for compact video-and-language representations. Adv. Neural. Inf. Process. Syst. 35, 30291\u201330306 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"30_CR10","doi-asserted-by":"publisher","first-page":"3367","DOI":"10.1109\/TIP.2023.3276570","volume":"32","author":"H Li","year":"2023","unstructured":"Li, H., Huang, J., Jin, P., Song, G., Wu, Q., Chen, J.: Weakly-supervised 3d spatial reasoning for text-based visual question answering. IEEE Trans. Image Process. 32, 3367\u20133382 (2023)","journal-title":"IEEE Trans. Image Process."},{"key":"30_CR11","doi-asserted-by":"crossref","unstructured":"Liang, T., Lin, G., Feng, L., Zhang, Y., Lv, F.: Attention is not enough: mitigating the distribution discrepancy in asynchronous multimodal sequence fusion. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8148\u20138156 (2021)","DOI":"10.1109\/ICCV48922.2021.00804"},{"issue":"2","key":"30_CR12","doi-asserted-by":"publisher","DOI":"10.1016\/j.ipm.2022.103229","volume":"60","author":"H Lin","year":"2023","unstructured":"Lin, H., Zhang, P., Ling, J., Yang, Z., Lee, L.K., Liu, W.: Ps-mixer: a polar-vector and strength-vector mixer model for multimodal sentiment analysis. Inf. Process. Manag. 60(2), 103229 (2023)","journal-title":"Inf. Process. Manag."},{"key":"30_CR13","doi-asserted-by":"crossref","unstructured":"Liu, Z., Shen, Y., Lakshminarasimhan, V.B., Liang, P.P., Zadeh, A., Morency, L.P.: Efficient low-rank multimodal fusion with modality-specific factors (2018). arXiv preprint arXiv:1806.00064","DOI":"10.18653\/v1\/P18-1209"},{"key":"30_CR14","doi-asserted-by":"crossref","unstructured":"Lv, F., Chen, X., Huang, Y., Duan, L., Lin, G.: Progressive modality reinforcement for human multimodal emotion recognition from unaligned multimodal sequences. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2554\u20132562 (2021)","DOI":"10.1109\/CVPR46437.2021.00258"},{"key":"30_CR15","unstructured":"Van\u00a0der Maaten, L., Hinton, G.: Visualizing data using t-SNE. J. Mach. Learn. Res. 9(11) (2008)"},{"key":"30_CR16","doi-asserted-by":"crossref","unstructured":"Mai, S., Hu, H., Xing, S.: Modality to modality translation: an adversarial representation learning and graph fusion network for multimodal fusion. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a034, pp. 164\u2013172 (2020)","DOI":"10.1609\/aaai.v34i01.5347"},{"key":"30_CR17","doi-asserted-by":"crossref","unstructured":"Michaud, F., Pirjanian, P., Audet, J., L\u00e9tourneau, D.: Artificial emotion and social robotics. In: Distributed Autonomous Robotic Systems, vol. 4, pp. 121\u2013130 (2000)","DOI":"10.1007\/978-4-431-67919-6_12"},{"key":"30_CR18","first-page":"14200","volume":"34","author":"A Nagrani","year":"2021","unstructured":"Nagrani, A., Yang, S., Arnab, A., Jansen, A., Schmid, C., Sun, C.: Attention bottlenecks for multimodal fusion. Adv. Neural. Inf. Process. Syst. 34, 14200\u201314213 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"30_CR19","doi-asserted-by":"crossref","unstructured":"Rahman, W., Hasan, M.K., Lee, S., Zadeh, A., Mao, C., Morency, L.P., Hoque, E.: Integrating multimodal information in large pretrained transformers. In: Proceedings of the Conference Association for Computational Linguistics Meeting, vol.\u00a02020, p.\u00a02359. NIH Public Access (2020)","DOI":"10.18653\/v1\/2020.acl-main.214"},{"key":"30_CR20","doi-asserted-by":"crossref","unstructured":"Sun, Z., Sarma, P., Sethares, W., Liang, Y.: Learning relationships between text, audio, and video via deep canonical correlation for multimodal language analysis. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a034, pp. 8992\u20138999 (2020)","DOI":"10.1609\/aaai.v34i05.6431"},{"key":"30_CR21","doi-asserted-by":"crossref","unstructured":"Tsai, Y.H.H., Bai, S., Liang, P.P., Kolter, J.Z., Morency, L.P., Salakhutdinov, R.: Multimodal transformer for unaligned multimodal language sequences. In: Proceedings of the Conference Association for Computational Linguistics Meeting, vol.\u00a02019, p.\u00a06558. NIH Public Access (2019)","DOI":"10.18653\/v1\/P19-1656"},{"key":"30_CR22","unstructured":"Tsai, Y.H.H., Liang, P.P., Zadeh, A., Morency, L.P., Salakhutdinov, R.: Learning factorized multimodal representations (2018). arXiv preprint arXiv:1806.06176"},{"key":"30_CR23","doi-asserted-by":"crossref","unstructured":"Wang, Y., Shen, Y., Liu, Z., Liang, P.P., Zadeh, A., Morency, L.P.: Words can shift: dynamically adjusting word representations using nonverbal behaviors. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a033, pp. 7216\u20137223 (2019)","DOI":"10.1609\/aaai.v33i01.33017216"},{"key":"30_CR24","doi-asserted-by":"crossref","unstructured":"Wei, Y., Yuan, S., Yang, R., Shen, L., Li, Z., Wang, L., Chen, M.: Tackling modality heterogeneity with multi-view calibration network for multimodal sentiment detection. In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics, vol. 1 (Long Papers), pp. 5240\u20135252 (2023)","DOI":"10.18653\/v1\/2023.acl-long.287"},{"key":"30_CR25","doi-asserted-by":"crossref","unstructured":"Wu, Y., Lin, Z., Zhao, Y., Qin, B., Zhu, L.N.: A text-centered shared-private framework via cross-modal prediction for multimodal sentiment analysis. In: Findings of the Association for Computational Linguistics: ACL-IJCNLP 2021, pp. 4730\u20134738 (2021)","DOI":"10.18653\/v1\/2021.findings-acl.417"},{"key":"30_CR26","unstructured":"Xia, Y., Huang, H., Zhu, J., Zhao, Z.: Achieving cross modal generalization with multimodal unified representation. Adv. Neural Inf. Process. Syst. 36 (2024)"},{"key":"30_CR27","doi-asserted-by":"crossref","unstructured":"Yang, D., Huang, S., Kuang, H., Du, Y., Zhang, L.: Disentangled representation learning for multimodal emotion recognition. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 1642\u20131651 (2022)","DOI":"10.1145\/3503161.3547754"},{"key":"30_CR28","doi-asserted-by":"crossref","unstructured":"Yang, J., Yu, Y., Niu, D., Guo, W., Xu, Y.: Confede: Contrastive feature decomposition for multimodal sentiment analysis. In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics, vol. 1 (Long Papers), pp. 7617\u20137630 (2023)","DOI":"10.18653\/v1\/2023.acl-long.421"},{"key":"30_CR29","doi-asserted-by":"crossref","unstructured":"Yu, W., Xu, H., Meng, F., Zhu, Y., Ma, Y., Wu, J., Zou, J., Yang, K.: Ch-sims: a Chinese multimodal sentiment analysis dataset with fine-grained annotation of modality. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, pp. 3718\u20133727 (2020)","DOI":"10.18653\/v1\/2020.acl-main.343"},{"key":"30_CR30","doi-asserted-by":"crossref","unstructured":"Zadeh, A., Chen, M., Poria, S., Cambria, E., Morency, L.P.: Tensor fusion network for multimodal sentiment analysis (2017). arXiv preprint arXiv:1707.07250","DOI":"10.18653\/v1\/D17-1115"},{"key":"30_CR31","doi-asserted-by":"crossref","unstructured":"Zadeh, A., Liang, P.P., Mazumder, N., Poria, S., Cambria, E., Morency, L.P.: Memory fusion network for multi-view sequential learning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a032 (2018)","DOI":"10.1609\/aaai.v32i1.12021"},{"key":"30_CR32","unstructured":"Zadeh, A., Zellers, R., Pincus, E., Morency, L.P.: Mosi: multimodal corpus of sentiment intensity and subjectivity analysis in online opinion videos (2016). arXiv preprint arXiv:1606.06259"},{"key":"30_CR33","unstructured":"Zadeh, A.B., Liang, P.P., Poria, S., Cambria, E., Morency, L.P.: Multimodal language analysis in the wild: Cmu-mosei dataset and interpretable dynamic fusion graph. In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics, vol. 1 (Long Papers), pp. 2236\u20132246 (2018)"},{"key":"30_CR34","unstructured":"Zellinger, W., Grubinger, T., Lughofer, E., Natschl\u00e4ger, T., Saminger-Platz, S.: Central moment discrepancy (cmd) for domain-invariant representation learning (2017). arXiv preprint arXiv:1702.08811"},{"key":"30_CR35","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Chen, M., Shen, J., Wang, C.: Tailor versatile multi-modal learning for multi-label emotion recognition. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a036, pp. 9100\u20139108 (2022)","DOI":"10.1609\/aaai.v36i8.20895"},{"key":"30_CR36","unstructured":"Zhu, B., Lin, B., Ning, M., Yan, Y., Cui, J., Wang, H., Pang, Y., Jiang, W., Zhang, J., Li, Z., et\u00a0al.: Languagebind: extending video-language pretraining to n-modality by language-based semantic alignment (2023). arXiv preprint arXiv:2310.01852"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition and Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-97-8502-5_30","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T14:21:48Z","timestamp":1730384508000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-97-8502-5_30"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,1]]},"ISBN":["9789819785018","9789819785025"],"references-count":36,"URL":"https:\/\/doi.org\/10.1007\/978-981-97-8502-5_30","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,1]]},"assertion":[{"value":"1 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"PRCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Chinese Conference on Pattern Recognition and Computer Vision  (PRCV)","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Urumqi","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18 October 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"7","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ccprcv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/2024.prcv.cn\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}