{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T07:42:51Z","timestamp":1763192571944,"version":"3.45.0"},"reference-count":46,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,6,30]]},"DOI":"10.1109\/ijcnn64981.2025.11227404","type":"proceedings-article","created":{"date-parts":[[2025,11,14]],"date-time":"2025-11-14T18:46:15Z","timestamp":1763145975000},"page":"1-8","source":"Crossref","is-referenced-by-count":0,"title":["Multi-Scale Sequence Fusion Model for Multimodal Sentiment Analysis: Leveraging Image-Text Interaction and Sequence Modeling"],"prefix":"10.1109","author":[{"given":"Yulei","family":"Zhang","sequence":"first","affiliation":[{"name":"Anhui University of Science and Technology,School of Computer Science and Engineering,Huainan,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guangli","family":"Zhu","sequence":"additional","affiliation":[{"name":"Anhui University of Science and Technology,School of Computer Science and Engineering,Huainan,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jixu","family":"Zhang","sequence":"additional","affiliation":[{"name":"Anhui University of Science and Technology,School of Computer Science and Engineering,Huainan,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiajia","family":"Liu","sequence":"additional","affiliation":[{"name":"Anhui University of Science and Technology,School of Computer Science and Engineering,Huainan,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chunqing","family":"Wang","sequence":"additional","affiliation":[{"name":"Anhui University of Science and Technology,School of Computer Science and Engineering,Huainan,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shunxiang","family":"Zhang","sequence":"additional","affiliation":[{"name":"Anhui University of Science and Technology,School of Computer Science and Engineering,Huainan,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/j.asoc.2023.111206"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2022.09.025"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2021.107018"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2021.3075573"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1016\/j.asej.2014.04.011"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2017.12.006"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/3586075"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2012.10.009"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1145\/3132847.3133142"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ISI.2017.8004895"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2023.110502"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2020.3035277"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2022.3160060"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.287"},{"key":"ref15","first-page":"328","article-title":"Multimodal sentiment detection based on multi-channel graph neural networks","volume-title":"Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics (ACL\u201921) and the 11th International Joint Conference on Natural Language Processing (IJCNLP\u201921)","author":"Yang"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1007\/s10489-023-05151-w"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2022.119240"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/3593583"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3094362"},{"article-title":"Vilt: Vision-and-language transformer without convolution or region supervision","volume-title":"International conference on machine learning","author":"Kim","key":"ref20"},{"journal-title":"Roberta: A robustly optimized bert pretraining approach","year":"2019","author":"Liu","key":"ref21"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00179"},{"article-title":"itransformer: Inverted transformers are effective for time series forecasting","year":"2023","author":"Liu","key":"ref24"},{"key":"ref25","first-page":"S0140525X16001837","article-title":"Attention Is All You Need.(Nips), 2017","volume":"10","author":"Vaswani","year":"2017"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N16-1030"},{"article-title":"Efficiently modeling long sequences with structured state spaces","year":"2021","author":"Gu","key":"ref27"},{"key":"ref28","first-page":"35971","article-title":"On the parameterization and ini-tialization of diagonal state space models","volume":"35","author":"Gu","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"article-title":"Hungry hungry hippos: Towards language modeling with state space models","year":"2022","author":"Fu","key":"ref29"},{"article-title":"Mamba: Linear-time sequence modeling with selective state spaces","year":"2023","author":"Gu","key":"ref30"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2024.128104"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2024.3521411"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1007\/s44267-024-00072-9"},{"article-title":"Vision Mamba: Efficient visual representation learning with bidirectional state space model","volume-title":"Int. Conf. on Mach. Learn","author":"Zhu","key":"ref34"},{"article-title":"Focal loss for dense object detection","volume-title":"proceedings of the IEEE conference on computer vision and pattern recognition.","author":"Ross","key":"ref35"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-27674-8_2"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1145\/3132847.3133142"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/isi.2017.8004895"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1145\/3209978.3210093"},{"key":"ref40","first-page":"328","article-title":"Multimodal sentiment detection based on multi-channel graph neural networks","volume-title":"Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics (ACL\u201921) and the 11th International Joint Conference on Natural Language Processing (IJCNLP\u201921).","author":"Yang"},{"key":"ref41","first-page":"2282","article-title":"CLMLF: A Contrastive Learning and Multi-Layer Fusion Method for Multimodal Sentiment Detection","volume-title":"Findings of the Association for Computational Linguistics (NAACL\u201922).","author":"Li"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2022.3160060"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2024.3405662"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.287"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/p19-1239"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1145\/3132847.3133142"}],"event":{"name":"2025 International Joint Conference on Neural Networks (IJCNN)","start":{"date-parts":[[2025,6,30]]},"location":"Rome, Italy","end":{"date-parts":[[2025,7,5]]}},"container-title":["2025 International Joint Conference on Neural Networks (IJCNN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11227166\/11227148\/11227404.pdf?arnumber=11227404","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T07:38:29Z","timestamp":1763192309000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11227404\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,30]]},"references-count":46,"URL":"https:\/\/doi.org\/10.1109\/ijcnn64981.2025.11227404","relation":{},"subject":[],"published":{"date-parts":[[2025,6,30]]}}}