{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T09:06:37Z","timestamp":1784279197292,"version":"3.55.0"},"reference-count":55,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"NSFC","doi-asserted-by":"publisher","award":["62225202"],"award-info":[{"award-number":["62225202"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"NSFC","doi-asserted-by":"publisher","award":["62302023"],"award-info":[{"award-number":["62302023"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.patcog.2026.113700","type":"journal-article","created":{"date-parts":[[2026,4,17]],"date-time":"2026-04-17T16:16:13Z","timestamp":1776442573000},"page":"113700","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PC","title":["MINA: Multimodal intention analysis of social media posts via LLM-guided audio-visual-text reasoning"],"prefix":"10.1016","volume":"179","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7013-6913","authenticated-orcid":false,"given":"Feihong","family":"Lu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tao","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ziqin","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yudi","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shiqi","family":"Gao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yangyifei","family":"Luo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zengxu","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qian","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qingyun","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5152-0055","authenticated-orcid":false,"given":"Jianxin","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patcog.2026.113700_b1","doi-asserted-by":"crossref","DOI":"10.1007\/978-3-642-69746-3_2","article-title":"From intentions to actions: A theory of planned behavior","author":"Ajzen","year":"1985","journal-title":"Action Control.: From Cogn. Behavior\/Springer"},{"key":"10.1016\/j.patcog.2026.113700_b2","first-page":"1","article-title":"Social media discourses as multimodal assemblages","author":"Simungala","year":"2025","journal-title":"J. Multicult. Discourses"},{"key":"10.1016\/j.patcog.2026.113700_b3","doi-asserted-by":"crossref","DOI":"10.3389\/fpsyg.2022.1046735","article-title":"Short video users\u2019 personality traits and social sharing motivation","volume":"13","author":"Da-Yong","year":"2022","journal-title":"Front. Psychol."},{"key":"10.1016\/j.patcog.2026.113700_b4","series-title":"International Semantic Web Conference","first-page":"121","article-title":"Rethinking uncertainly missing and ambiguous visual modality in multi-modal entity alignment","author":"Chen","year":"2023"},{"key":"10.1016\/j.patcog.2026.113700_b5","unstructured":"A. Radford, J.W. Kim, C. Hallacy, A. Ramesh, G. Goh, S. Agarwal, G. Sastry, A. Askell, P. Mishkin, J. Clark, et al., Learning transferable visual models from natural language supervision, in: International Conference on Machine Learning, 2021, pp. 8748\u20138763."},{"key":"10.1016\/j.patcog.2026.113700_b6","doi-asserted-by":"crossref","unstructured":"W. Wang, T. Fang, C. Li, H. Shi, et al., CANDLE: Iterative Conceptualization and Instantiation Distillation from Large Language Models for Commonsense Reasoning, in: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), 2024, pp. 2351\u20132374.","DOI":"10.18653\/v1\/2024.acl-long.128"},{"key":"10.1016\/j.patcog.2026.113700_b7","doi-asserted-by":"crossref","unstructured":"F. Lu, W. Wang, Y. Luo, Z. Zhu, Q. Sun, B. Xu, H. Shi, S. Gao, Q. Li, Y. Song, et al., Miko: Multimodal intention knowledge distillation from large language models for social-media commonsense discovery, in: Proceedings of the 32nd ACM International Conference on Multimedia, 2024, pp. 3303\u20133312.","DOI":"10.1145\/3664647.3681339"},{"key":"10.1016\/j.patcog.2026.113700_b8","series-title":"Findings of the Association for Computational Linguistics: ACL 2023","first-page":"1173","article-title":"FolkScope: Intention knowledge graph construction for E-commerce commonsense discovery","author":"Yu","year":"2023"},{"key":"10.1016\/j.patcog.2026.113700_b9","doi-asserted-by":"crossref","unstructured":"B. Xu, W. Wang, H. Shi, W. Ding, H. Jing, T. Fang, J. Bai, X. Liu, C. Yu, Z. Li, et al., Mind: Multimodal shopping intention distillation from large vision-language models for e-commerce purchase understanding, in: Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, 2024, pp. 7800\u20137815.","DOI":"10.18653\/v1\/2024.emnlp-main.446"},{"issue":"1","key":"10.1016\/j.patcog.2026.113700_b10","doi-asserted-by":"crossref","first-page":"38","DOI":"10.1016\/j.pragma.2012.12.010","article-title":"A multimodal analysis of facework strategies in a corpus of charity ads on british television","volume":"49","author":"Pennock-Speck","year":"2013","journal-title":"J. Pragmat."},{"issue":"19","key":"10.1016\/j.patcog.2026.113700_b11","doi-asserted-by":"crossref","first-page":"6077","DOI":"10.1080\/10447318.2023.2247606","article-title":"MIUIC: A human-computer collaborative multimodal intention-understanding algorithm incorporating comfort analysis","volume":"40","author":"Zhou","year":"2024","journal-title":"Int. J. Human\u2013Comput. Interact."},{"key":"10.1016\/j.patcog.2026.113700_b12","unstructured":"H. Zhang, X. Wang, H. Xu, Q. Zhou, K. Gao, J. Su, J. Zhao, W. Li, Y. Chen, MIntRec2.0: A Large-scale Benchmark Dataset for Multimodal Intent Recognition and Out-of-scope Detection in Conversations, in: The Twelfth International Conference on Learning Representations, 2024."},{"key":"10.1016\/j.patcog.2026.113700_b13","series-title":"2023 IEEE International Conference on Multimedia and Expo","first-page":"2825","article-title":"Multimodal fake news detection via clip-guided learning","author":"Zhou","year":"2023"},{"key":"10.1016\/j.patcog.2026.113700_b14","first-page":"17114","article-title":"Token-level contrastive learning with modality-aware prompting for multimodal intent recognition","volume":"vol. 38","author":"Zhou","year":"2024"},{"key":"10.1016\/j.patcog.2026.113700_b15","first-page":"35254","article-title":"Twibot-22: Towards graph-based twitter bot detection","volume":"35","author":"Feng","year":"2022","journal-title":"NeurIPS"},{"key":"10.1016\/j.patcog.2026.113700_b16","doi-asserted-by":"crossref","unstructured":"K. Sun, Z. Xie, M. Ye, H. Zhang, Contextual augmented global contrast for multimodal intent recognition, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 26963\u201326973.","DOI":"10.1109\/CVPR52733.2024.02546"},{"key":"10.1016\/j.patcog.2026.113700_b17","article-title":"Unlocking human intent perception through multimodal large models","author":"Wang","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113700_b18","doi-asserted-by":"crossref","unstructured":"Q. Yang, Q. Shi, T. Wang, M. Ye, Uncertain Multimodal Intention and Emotion Understanding in the Wild, in: Proceedings of the Computer Vision and Pattern Recognition Conference, 2025, pp. 24700\u201324709.","DOI":"10.1109\/CVPR52734.2025.02300"},{"key":"10.1016\/j.patcog.2026.113700_b19","series-title":"Distilling the knowledge in a neural network","author":"Hinton","year":"2015"},{"key":"10.1016\/j.patcog.2026.113700_b20","first-page":"29545","article-title":"Multimodal commonsense knowledge distillation for visual question answering (student abstract)","volume":"vol. 39","author":"Yang","year":"2025"},{"key":"10.1016\/j.patcog.2026.113700_b21","doi-asserted-by":"crossref","DOI":"10.1016\/j.neunet.2024.106272","article-title":"Layerwised multimodal knowledge distillation for vision-language pretrained model","volume":"175","author":"Wang","year":"2024","journal-title":"Neural Netw."},{"issue":"4","key":"10.1016\/j.patcog.2026.113700_b22","doi-asserted-by":"crossref","first-page":"1869","DOI":"10.1109\/TCSS.2024.3402270","article-title":"Zero-shot cross-lingual knowledge transfer in vqa via multimodal distillation","volume":"12","author":"Weng","year":"2024","journal-title":"IEEE Trans. Comput. Soc. Syst."},{"key":"10.1016\/j.patcog.2026.113700_b23","unstructured":"F. Shu, Y. Liao, L. Zhang, L. Zhuo, C. Xu, G. Zhang, et al., LLaVA-MoD: Making LLaVA Tiny via MoE-Knowledge Distillation, in: The Thirteenth International Conference on Learning Representations, 2025."},{"key":"10.1016\/j.patcog.2026.113700_b24","doi-asserted-by":"crossref","unstructured":"Q. Feng, W. Li, T. Lin, X. Chen, Align-KD: Distilling Cross-Modal Alignment Knowledge for Mobile Vision-Language Large Model Enhancement, in: Proceedings of the Computer Vision and Pattern Recognition Conference, 2025, pp. 4178\u20134188.","DOI":"10.1109\/CVPR52734.2025.00395"},{"key":"10.1016\/j.patcog.2026.113700_b25","unstructured":"T. Yang, Y. Hu, F. Lu, Z. Zhang, Q. Sun, J. Li, BotUmc: An Uncertainty-Aware Twitter Bot Detection with Multi-view Causal Inference, 2025, arXiv preprint arXiv:2503.03775."},{"key":"10.1016\/j.patcog.2026.113700_b26","article-title":"Distilling implicit multimodal knowledge into large language models for zero-resource dialogue generation","author":"Zhang","year":"2025","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.patcog.2026.113700_b27","series-title":"Qwen2. 5-vl technical report","author":"Bai","year":"2025"},{"key":"10.1016\/j.patcog.2026.113700_b28","doi-asserted-by":"crossref","unstructured":"H. Liu, C. Li, Y. Li, Y.J. Lee, Improved baselines with visual instruction tuning, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 26296\u201326306.","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"10.1016\/j.patcog.2026.113700_b29","series-title":"OSUM: Advancing open speech understanding models with limited resources in academia","author":"Geng","year":"2025"},{"key":"10.1016\/j.patcog.2026.113700_b30","series-title":"InternVideo2.5: Empowering video MLLMs with long and rich context modeling","author":"Wang","year":"2025"},{"key":"10.1016\/j.patcog.2026.113700_b31","doi-asserted-by":"crossref","unstructured":"H. Xu, G. Ghosh, P.-Y. Huang, D. Okhonko, A. Aghajanyan, F. Metze, L. Zettlemoyer, C. Feichtenhofer, VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding, in: Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, 2021, pp. 6787\u20136800.","DOI":"10.18653\/v1\/2021.emnlp-main.544"},{"key":"10.1016\/j.patcog.2026.113700_b32","first-page":"119336","article-title":"Streaming long video understanding with large language models","volume":"37","author":"Qian","year":"2024","journal-title":"NeurIPS"},{"key":"10.1016\/j.patcog.2026.113700_b33","series-title":"AAAI","first-page":"3027","article-title":"ATOMIC: an atlas of machine commonsense for if-then reasoning","author":"Sap","year":"2019"},{"issue":"45","key":"10.1016\/j.patcog.2026.113700_b34","doi-asserted-by":"crossref","DOI":"10.1073\/pnas.2405460121","article-title":"Evaluating large language models in theory of mind tasks","volume":"121","author":"Kosinski","year":"2024","journal-title":"Proc. Natl. Acad. Sci."},{"key":"10.1016\/j.patcog.2026.113700_b35","doi-asserted-by":"crossref","unstructured":"W. Wu, H. Li, H. Wang, K.Q. Zhu, Probase: A probabilistic taxonomy for text understanding, in: Proceedings of the 2012 ACM SIGMOD International Conference on Management of Data, 2012, pp. 481\u2013492.","DOI":"10.1145\/2213836.2213891"},{"issue":"8081","key":"10.1016\/j.patcog.2026.113700_b36","doi-asserted-by":"crossref","first-page":"633","DOI":"10.1038\/s41586-025-09422-z","article-title":"DeepSeek-R1 incentivizes reasoning in LLMs through reinforcement learning","volume":"645","author":"Guo","year":"2025","journal-title":"Nature"},{"key":"10.1016\/j.patcog.2026.113700_b37","series-title":"Llama 2: Open foundation and fine-tuned chat models","author":"Touvron","year":"2023"},{"key":"10.1016\/j.patcog.2026.113700_b38","series-title":"The llama 3 herd of models","author":"Grattafiori","year":"2024"},{"key":"10.1016\/j.patcog.2026.113700_b39","series-title":"Mistral 7B","author":"Jiang","year":"2024"},{"key":"10.1016\/j.patcog.2026.113700_b40","series-title":"Qwen2 technical report","author":"Team","year":"2024"},{"key":"10.1016\/j.patcog.2026.113700_b41","series-title":"Qwen2.5: A party of foundation models","author":"Qwen Team","year":"2024"},{"key":"10.1016\/j.patcog.2026.113700_b42","series-title":"Qwen3 technical report","author":"Yang","year":"2025"},{"key":"10.1016\/j.patcog.2026.113700_b43","unstructured":"D. Zhu, J. Chen, X. Shen, X. Li, M. Elhoseiny, MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models, in: The Twelfth International Conference on Learning Representations, 2024."},{"key":"10.1016\/j.patcog.2026.113700_b44","series-title":"Gpt-4 technical report","author":"Achiam","year":"2023"},{"key":"10.1016\/j.patcog.2026.113700_b45","series-title":"Doubao-1.5-pro","author":"Team","year":"2025"},{"key":"10.1016\/j.patcog.2026.113700_b46","unstructured":"P. Velickovic, G. Cucurull, A. Casanova, A. Romero, P. Li\u00f2, Y. Bengio, Graph Attention Networks, in: The 6th International Conference on Learning Representations, 2018."},{"issue":"5\u20136","key":"10.1016\/j.patcog.2026.113700_b47","doi-asserted-by":"crossref","first-page":"602","DOI":"10.1016\/j.neunet.2005.06.042","article-title":"Framewise phoneme classification with bidirectional LSTM and other neural network architectures","volume":"18","author":"Graves","year":"2005","journal-title":"Neural Netw."},{"key":"10.1016\/j.patcog.2026.113700_b48","doi-asserted-by":"crossref","unstructured":"S. Feng, H. Wan, N. Wang, M. Luo, BotRGCN: Twitter bot detection with relational graph convolutional networks, in: Proceedings of the 2021 IEEE\/ACM International Conference on Advances in Social Networks Analysis and Mining, 2021, pp. 236\u2013239.","DOI":"10.1145\/3487351.3488336"},{"key":"10.1016\/j.patcog.2026.113700_b49","doi-asserted-by":"crossref","unstructured":"T. Xiong, P. Zhang, H. Zhu, Y. Yang, Sarcasm detection with self-matching networks and low-rank bilinear pooling, in: The World Wide Web Conference, 2019, pp. 2115\u20132124.","DOI":"10.1145\/3308558.3313735"},{"key":"10.1016\/j.patcog.2026.113700_b50","series-title":"Roberta: A robustly optimized bert pretraining approach","author":"Liu","year":"2019"},{"key":"10.1016\/j.patcog.2026.113700_b51","unstructured":"P. He, X. Liu, J. Gao, W. Chen, Deberta: decoding-Enhanced Bert with Disentangled Attention, in: The 9th International Conference on Learning Representations, 2021."},{"key":"10.1016\/j.patcog.2026.113700_b52","doi-asserted-by":"crossref","unstructured":"Y. Liu, Z. Tan, H. Wang, S. Feng, Q. Zheng, M. Luo, Botmoe: Twitter bot detection with community-aware mixtures of modal-specific experts, in: Proceedings of the 46th International ACM SIGIR Conference on Research and Development in Information Retrieval, 2023, pp. 485\u2013495.","DOI":"10.1145\/3539618.3591646"},{"key":"10.1016\/j.patcog.2026.113700_b53","series-title":"ACL","first-page":"2506","article-title":"Multi-modal sarcasm detection in Twitter with hierarchical fusion model","author":"Cai","year":"2019"},{"key":"10.1016\/j.patcog.2026.113700_b54","first-page":"3977","article-title":"Heterogeneity-aware twitter bot detection with relational graph transformers","volume":"vol. 36","author":"Feng","year":"2022"},{"key":"10.1016\/j.patcog.2026.113700_b55","series-title":"EMNLP","first-page":"4995","article-title":"Towards multi-modal sarcasm detection via hierarchical congruity modeling with knowledge enhancement","author":"Liu","year":"2022"}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326006655?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326006655?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T08:31:26Z","timestamp":1784277086000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326006655"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":55,"alternative-id":["S0031320326006655"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113700","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"MINA: Multimodal intention analysis of social media posts via LLM-guided audio-visual-text reasoning","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113700","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"113700"}}