{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,12]],"date-time":"2026-07-12T03:23:49Z","timestamp":1783826629571,"version":"3.55.0"},"reference-count":36,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2025,7,7]],"date-time":"2025-07-07T00:00:00Z","timestamp":1751846400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,7,7]],"date-time":"2025-07-07T00:00:00Z","timestamp":1751846400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"National Students\u2019 Platform Innovation and Entrepreneurship Training Program Support Project","award":["02410300260E"],"award-info":[{"award-number":["02410300260E"]}]},{"name":"National Students\u2019 Platform Innovation and Entrepreneurship Training Program Support Project","award":["02410300260E"],"award-info":[{"award-number":["02410300260E"]}]},{"name":"National Students\u2019 Platform Innovation and Entrepreneurship Training Program Support Project","award":["02410300260E"],"award-info":[{"award-number":["02410300260E"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2025,8]]},"DOI":"10.1007\/s00530-025-01895-2","type":"journal-article","created":{"date-parts":[[2025,7,7]],"date-time":"2025-07-07T09:04:15Z","timestamp":1751879055000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Enhancing semantics consistency via hybrid attention fusion in multimodal sentiment analysis of short videos"],"prefix":"10.1007","volume":"31","author":[{"given":"Xuanchi","family":"Gong","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ziyang","family":"Xue","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tengjun","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yubao","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yifan","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,7,7]]},"reference":[{"key":"1895_CR1","unstructured":"Yang, S., Zhao, Y., Ma, Y.: Analysis of the reasons and development of short video application-taking tik tok as an example. In Proceedings of the 2019 9th International Conference on Information and Social Science (ICISS 2019), Manila, Philippines, pages 12\u201314, (2019)"},{"issue":"18","key":"1895_CR2","doi-asserted-by":"publisher","first-page":"6681","DOI":"10.3390\/ijerph17186681","volume":"17","author":"F Peihua","year":"2020","unstructured":"Peihua, F., Jing, B., Chen, T., Yang, J., Cong, G.: Modeling network public opinion propagation with the consideration of individual emotions. Int. J. Environ. Res. Public Health 17(18), 6681 (2020)","journal-title":"Int. J. Environ. Res. Public Health"},{"key":"1895_CR3","unstructured":"Wei, Q., Zhou, Y., Xiao, L., Zhang, Y.: Mseva: A system for multimodal short videos emotion visual analysis. arXiv preprint arXiv:2312.04279, (2023)"},{"key":"1895_CR4","doi-asserted-by":"crossref","unstructured":"Zadeh, A., Chen, M., Poria, S., Cambria, E., Morency, L.-P.: Tensor fusion network for multimodal sentiment analysis. arXiv preprint arXiv:1707.07250, (2017)","DOI":"10.18653\/v1\/D17-1115"},{"key":"1895_CR5","doi-asserted-by":"crossref","unstructured":"Hazarika, D., Zimmermann, R., Poria, S.: Misa: Modality-invariant and-specific representations for multimodal sentiment analysis. In Proceedings of the 28th ACM international conference on multimedia, pages 1122\u20131131, (2020)","DOI":"10.1145\/3394171.3413678"},{"key":"1895_CR6","doi-asserted-by":"publisher","first-page":"10790","DOI":"10.1609\/aaai.v35i12.17289","volume":"35","author":"Yu Wenmeng","year":"2021","unstructured":"Wenmeng, Yu., Hua, X., Yuan, Z., Jiele, W.: Learning modality-specific representations with self-supervised multi-task learning for multimodal sentiment analysis. In Proceedings of the AAAI conference on artificial intelligence 35, 10790\u201310797 (2021)","journal-title":"In Proceedings of the AAAI conference on artificial intelligence"},{"key":"1895_CR7","doi-asserted-by":"crossref","unstructured":"Ghosal, D., Majumder, N., Gelbukh, A., Mihalcea, R., Poria, S.: Cosmic: Commonsense knowledge for emotion identification in conversations. arXiv preprint arXiv:2010.02795, (2020)","DOI":"10.18653\/v1\/2020.findings-emnlp.224"},{"issue":"5","key":"1895_CR8","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3465055","volume":"12","author":"S Chaudhari","year":"2021","unstructured":"Chaudhari, S., Mithal, V., Polatkan, G., Ramanath, R.: An attentive survey of attention models. ACM Transactions on Intelligent Systems and Technology (TIST) 12(5), 1\u201332 (2021)","journal-title":"ACM Transactions on Intelligent Systems and Technology (TIST)"},{"key":"1895_CR9","doi-asserted-by":"crossref","unstructured":"Tsai, Y.-H. H., Bai, S., Liang, P. P., Kolter, J. Z., Morency, L.-P., Salakhutdinov, R.: Multimodal transformer for unaligned multimodal language sequences. In Proceedings of the conference. Association for computational linguistics. Meeting, volume 2019, page 6558. NIH Public Access, (2019)","DOI":"10.18653\/v1\/P19-1656"},{"issue":"4","key":"1895_CR10","first-page":"46","volume":"7","author":"Yu Zhang","year":"2023","unstructured":"Zhang, Yu., Haijun, Z., Yaqing, L., Kejin, L., Yueyang, W.: Multimodal sentiment analysis based on bidirectional mask attention mechanism. Data Analysis and Knowledge Discovery 7(4), 46\u201355 (2023)","journal-title":"Data Analysis and Knowledge Discovery"},{"key":"1895_CR11","doi-asserted-by":"crossref","unstructured":"Lea, C., Vidal, R., Reiter, A., Hager, G. D.:. Temporal convolutional networks: A unified approach to action segmentation. In Computer Vision\u2013ECCV 2016 Workshops: Amsterdam, The Netherlands, October 8-10 and 15-16, 2016, Proceedings, Part III 14, pages 47\u201354. Springer, (2016)","DOI":"10.1007\/978-3-319-49409-8_7"},{"key":"1895_CR12","unstructured":"Wu, X., Sun, H., Xue, J., Zhai, R., Kong, X., Nie, J., He, L.: emotions: A large-scale dataset for emotion recognition in short videos. arXiv preprint arXiv:2311.17335, (2023)"},{"key":"1895_CR13","doi-asserted-by":"publisher","DOI":"10.1016\/j.sasc.2024.200148","volume":"6","author":"H Shi","year":"2024","unstructured":"Shi, H.: A short video sentiment analysis model based on multimodal feature fusion. Systems and Soft Computing 6, 200148 (2024)","journal-title":"Systems and Soft Computing"},{"key":"1895_CR14","unstructured":"Tadas, B., Amir, Z., Chong, L. Y., Philippe, L.-M.: Openface 2.0: Facial behavior analysis toolkit. In 13th IEEE International Conference on Automatic Face & Gesture Recognition, (2018)"},{"key":"1895_CR15","doi-asserted-by":"crossref","unstructured":"Feng, Y., Gao, J., Xu, C.: Spatiotemporal orthogonal projection capsule network for incremental few-shot action recognition. IEEE Transactions on Multimedia, (2024)","DOI":"10.1109\/TMM.2024.3399453"},{"key":"1895_CR16","doi-asserted-by":"crossref","unstructured":"Dey, A., Biswas, S., et\u00a0al.: Workout action recognition in video streams using an attention driven residual dc-gru network. Computers, Materials & Continua, 79(2), (2024)","DOI":"10.32604\/cmc.2024.049512"},{"key":"1895_CR17","first-page":"38571","volume":"35","author":"X Yufei","year":"2022","unstructured":"Yufei, X., Zhang, J., Zhang, Q., Tao, D.: Vitpose: Simple vision transformer baselines for human pose estimation. Adv. Neural. Inf. Process. Syst. 35, 38571\u201338584 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1895_CR18","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., Zitnick, C. L.: Microsoft coco: Common objects in context. In Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13, pages 740\u2013755. Springer, (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"1895_CR19","unstructured":"Devlin, J.: Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805, (2018)"},{"key":"1895_CR20","unstructured":"Koroteev, M. V.: Bert: a review of applications in natural language processing and understanding. arXiv preprint arXiv:2103.11943, (2021)"},{"key":"1895_CR21","doi-asserted-by":"crossref","unstructured":"Degottex, G., Kane, J., Drugman, T., Raitio, T., Scherer, S.: Covarepa collaborative voice analysis repository for speech technologies. In 2014 ieee international conference on acoustics, speech and signal processing (icassp), pages 960\u2013964. IEEE, (2014)","DOI":"10.1109\/ICASSP.2014.6853739"},{"issue":"4","key":"1895_CR22","doi-asserted-by":"publisher","first-page":"195","DOI":"10.1007\/s00530-024-01402-z","volume":"30","author":"P Gao","year":"2024","unstructured":"Gao, P., Tao, C., Guan, D.: Fef-net: feature enhanced fusion network with crossmodal attention for multimodal humor prediction. Multimedia Syst. 30(4), 195 (2024)","journal-title":"Multimedia Syst."},{"key":"1895_CR23","unstructured":"Zellinger, W., Grubinger, T., Lughofer, E., Natschl\u00e4ger, T., Saminger-Platz, S.: Central moment discrepancy (cmd) for domain-invariant representation learning. arXiv preprint arXiv:1702.08811, (2017)"},{"key":"1895_CR24","unstructured":"Gao, J., Chen, M., Xiang, L., Xu, C.: A comprehensive survey on evidential deep learning and its applications. arXiv preprint arXiv:2409.04720, (2024)"},{"key":"1895_CR25","doi-asserted-by":"crossref","unstructured":"Gao, J., Chen, M., Xu, C.: Learning probabilistic presence-absence evidence for weakly-supervised audio-visual event perception. IEEE Transactions on Pattern Analysis and Machine Intelligence, (2025)","DOI":"10.1109\/TPAMI.2025.3546312"},{"key":"1895_CR26","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2021.107676","volume":"235","author":"W Ting","year":"2022","unstructured":"Ting, W., Peng, J., Zhang, W., Zhang, H., Tan, S., Yi, F., Ma, C., Huang, Y.: Video sentiment analysis with bimodal information-augmented multi-head attention. Knowl.-Based Syst. 235, 107676 (2022)","journal-title":"Knowl.-Based Syst."},{"key":"1895_CR27","doi-asserted-by":"crossref","unstructured":"Yu, W., Xu, H., Meng, F., Zhu, Y., Ma, Y., Wu, J., Zou, J., Yang, K.: Ch-sims: A chinese multimodal sentiment analysis dataset with fine-grained annotation of modality. In Proceedings of the 58th annual meeting of the association for computational linguistics, 3718\u20133727, (2020)","DOI":"10.18653\/v1\/2020.acl-main.343"},{"key":"1895_CR28","unstructured":"Zadeh, A., Zellers, R., Pincus, E., Morency, L.-P.: Mosi: multimodal corpus of sentiment intensity and subjectivity analysis in online opinion videos. arXiv preprint arXiv:1606.06259, (2016)"},{"key":"1895_CR29","unstructured":"Zadeh, A. B., Liang, P. P., Poria, S.: Erik Cambria, and Louis-Philippe Morency. Multimodal language analysis in the wild: Cmu-mosei dataset and interpretable dynamic fusion graph. In Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pages 2236\u20132246, (2018)"},{"key":"1895_CR30","doi-asserted-by":"crossref","unstructured":"Rahman, W., Hasan, M. K., Lee, S., Zadeh, A., Mao, C., Morency, L.-P, Hoque, E.: Integrating multimodal information in large pretrained transformers. In Proceedings of the conference. Association for Computational Linguistics. Meeting, volume 2020, page 2359. NIH Public Access, (2020)","DOI":"10.18653\/v1\/2020.acl-main.214"},{"issue":"1","key":"1895_CR31","doi-asserted-by":"publisher","first-page":"10","DOI":"10.1007\/s00530-023-01208-5","volume":"30","author":"Y Luo","year":"2024","unstructured":"Luo, Y., Rui, W., Liu, J., Tang, X.: Balanced sentimental information via multimodal interaction model. Multimedia Syst. 30(1), 10 (2024)","journal-title":"Multimedia Syst."},{"issue":"4","key":"1895_CR32","doi-asserted-by":"publisher","first-page":"228","DOI":"10.1007\/s00530-024-01421-w","volume":"30","author":"Q Huang","year":"2024","unstructured":"Huang, Q., Chen, J., Huang, C., Huang, X., Wang, Y.: Text-centered cross-sample fusion network for multimodal sentiment analysis. Multimedia Syst. 30(4), 228 (2024)","journal-title":"Multimedia Syst."},{"key":"1895_CR33","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2023.111346","volume":"285","author":"J Huang","year":"2024","unstructured":"Huang, J., Zhou, J., Tang, Z., Lin, J., Chen, C.Y.-C.: Tmbl: Transformer-based multimodal binding learning model for multimodal sentiment analysis. Knowl.-Based Syst. 285, 111346 (2024)","journal-title":"Knowl.-Based Syst."},{"key":"1895_CR34","doi-asserted-by":"crossref","unstructured":"Wu, Z., Gong, Z., Koo, J., Hirschberg, J.: Multimodal multi-loss fusion network for sentiment analysis. In Kevin Duh, Helena Gomez, and Steven Bethard, editors, Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), pages 3588\u20133602, Mexico City, Mexico. Association for Computational Linguistics (2024)","DOI":"10.18653\/v1\/2024.naacl-long.197"},{"key":"1895_CR35","unstructured":"Byun, S.-Y., Lee, W.: Vit-reciprocam: Gradient and attention-free visual explanations for vision transformer. arXiv preprint arXiv:2310.02588, (2023)"},{"key":"1895_CR36","unstructured":"Radford, A., Kim, J. W., Xu, T., Brockman, G., McLeavey, C., Sutskever, I.: Robust speech recognition via large-scale weak supervision. In International conference on machine learning, pages 28492\u201328518. PMLR, (2023)"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-01895-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-025-01895-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-01895-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,15]],"date-time":"2025-09-15T09:02:41Z","timestamp":1757926961000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-025-01895-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,7]]},"references-count":36,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2025,8]]}},"alternative-id":["1895"],"URL":"https:\/\/doi.org\/10.1007\/s00530-025-01895-2","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,7,7]]},"assertion":[{"value":"3 December 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 June 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 July 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"301"}}