{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,9]],"date-time":"2026-06-09T03:00:23Z","timestamp":1780974023040,"version":"3.54.1"},"reference-count":50,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100021171","name":"Basic and Applied Basic Research Foundation of Guangdong Province","doi-asserted-by":"publisher","award":["2025A1515010454"],"award-info":[{"award-number":["2025A1515010454"]}],"id":[{"id":"10.13039\/501100021171","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100021171","name":"Basic and Applied Basic Research Foundation of Guangdong Province","doi-asserted-by":"publisher","award":["2022A1515011835"],"award-info":[{"award-number":["2022A1515011835"]}],"id":[{"id":"10.13039\/501100021171","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62206314"],"award-info":[{"award-number":["62206314"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004000","name":"Guangzhou Municipal Science and Technology Program key projects","doi-asserted-by":"publisher","award":["2024A04J4388"],"award-info":[{"award-number":["2024A04J4388"]}],"id":[{"id":"10.13039\/501100004000","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1016\/j.eswa.2026.132076","type":"journal-article","created":{"date-parts":[[2026,3,16]],"date-time":"2026-03-16T17:08:52Z","timestamp":1773680932000},"page":"132076","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Multimodal feature disentangle-fusion network for detecting and grounding multi-modal media manipulation"],"prefix":"10.1016","volume":"319","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0663-199X","authenticated-orcid":false,"given":"Jinghui","family":"Qin","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-2443-487X","authenticated-orcid":false,"given":"Jialin","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5848-5624","authenticated-orcid":false,"given":"Tianshui","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8336-5109","authenticated-orcid":false,"given":"Zhijing","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.132076_bib0001","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2025.128571","article-title":"Fame: A lightweight spatio-temporal network for model attribution of face-swap deepfakes","volume":"292","author":"Ahmad","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132076_bib0002","series-title":"International conference on intelligent computing","first-page":"368","article-title":"Catching inter-modal artifacts: A cross-modal framework for temporal forgery localization","author":"Cai","year":"2025"},{"key":"10.1016\/j.eswa.2026.132076_bib0003","series-title":"Proceedings of the ACM web conference 2022","first-page":"2897","article-title":"Cross-modal ambiguity learning for multimodal fake news detection","author":"Chen","year":"2022"},{"key":"10.1016\/j.eswa.2026.132076_bib0004","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2025.127407","article-title":"Multi-view mutual learning network for multimodal fake news detection","volume":"279","author":"Cui","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132076_bib0005","series-title":"International conference on learning representations","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2026"},{"issue":"11","key":"10.1016\/j.eswa.2026.132076_bib0006","doi-asserted-by":"crossref","first-page":"139","DOI":"10.1145\/3422622","article-title":"Generative adversarial networks","volume":"63","author":"Goodfellow","year":"2020","journal-title":"Communications of the ACM"},{"key":"10.1016\/j.eswa.2026.132076_bib0007","first-page":"6840","article-title":"Denoising diffusion probabilistic models","volume":"33","author":"Ho","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132076_bib0008","series-title":"Proceedings of the 33rd ACM international conference on multimedia","first-page":"2362","article-title":"PgM: Partitioner guided modal learning framework","author":"Hu","year":"2025"},{"key":"10.1016\/j.eswa.2026.132076_bib0009","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"7132","article-title":"Squeeze-and-excitation networks","author":"Hu","year":"2018"},{"key":"10.1016\/j.eswa.2026.132076_bib0010","series-title":"Proceedings of NAACL-HLT","first-page":"2","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","volume":"1","author":"Kenton","year":"2019"},{"key":"10.1016\/j.eswa.2026.132076_bib0011","series-title":"The world wide web conference","first-page":"2915","article-title":"MVAE: Multimodal variational autoencoder for fake news detection","author":"Khattar","year":"2019"},{"key":"10.1016\/j.eswa.2026.132076_bib0012","series-title":"International conference on machine learning","first-page":"5583","article-title":"ViLT: Vision-and-language transformer without convolution or region supervision","author":"Kim","year":"2021"},{"issue":"1","key":"10.1016\/j.eswa.2026.132076_bib0013","doi-asserted-by":"crossref","first-page":"79","DOI":"10.1214\/aoms\/1177729694","article-title":"On information and sufficiency","volume":"22","author":"Kullback","year":"1951","journal-title":"The Annals of Mathematical Statistics"},{"key":"10.1016\/j.eswa.2026.132076_bib0014","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"5001","article-title":"Face X-ray for more general face forgery detection","author":"Li","year":"2020"},{"key":"10.1016\/j.eswa.2026.132076_bib0015","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2023.102037","article-title":"Towards multimodal disinformation detection by vision-language knowledge interaction","volume":"102","author":"Li","year":"2024","journal-title":"Information Fusion"},{"key":"10.1016\/j.eswa.2026.132076_bib0016","first-page":"1","article-title":"Unified frequency-assisted transformer framework for detecting and grounding multi-modal manipulation","volume":"133","author":"Liu","year":"2024","journal-title":"International Journal of Computer Vision"},{"key":"10.1016\/j.eswa.2026.132076_bib0017","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"448","article-title":"Detecting images generated by deep diffusion models using their local intrinsic dimensionality","author":"Lorenz","year":"2023"},{"key":"10.1016\/j.eswa.2026.132076_bib0018","unstructured":"Loshchilov, I. (2017). Decoupled weight decay regularization. arXiv: 1711.05101."},{"key":"10.1016\/j.eswa.2026.132076_bib0019","series-title":"International conference on learning representations","article-title":"Learning sparse neural networks through l_0 regularization","author":"Louizos","year":"2018"},{"key":"10.1016\/j.eswa.2026.132076_bib0020","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"16317","article-title":"Generalizing face forgery detection with high-frequency features","author":"Luo","year":"2021"},{"key":"10.1016\/j.eswa.2026.132076_bib0021","series-title":"2023 International conference on multimedia analysis and pattern recognition (MAPR)","first-page":"1","article-title":"Unmasking the artist: Discriminating human-drawn and AI-generated human face art through facial feature analysis","author":"Nguyen","year":"2023"},{"key":"10.1016\/j.eswa.2026.132076_bib0022","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.109863","article-title":"Few-shot forgery detection via guided adversarial interpolation","volume":"144","author":"Qiu","year":"2023","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.132076_bib0023","series-title":"International conference on machine learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"issue":"8","key":"10.1016\/j.eswa.2026.132076_bib0024","first-page":"9","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI blog"},{"key":"10.1016\/j.eswa.2026.132076_bib0025","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"658","article-title":"Generalized intersection over union: A metric and a loss for bounding box regression","author":"Rezatofighi","year":"2019"},{"key":"10.1016\/j.eswa.2026.132076_bib0026","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2025.127572","article-title":"SGDM-GRU: Spectral graph deep learning based gated recurrent unit model for accurate fake news detection","volume":"281","author":"Sahi","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132076_bib0027","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"6904","article-title":"Detecting and grounding multi-modal media manipulation","author":"Shao","year":"2023"},{"key":"10.1016\/j.eswa.2026.132076_bib0028","doi-asserted-by":"crossref","DOI":"10.1109\/TPAMI.2024.3367749","article-title":"Detecting and grounding multi-modal media manipulation and beyond","volume":"46","author":"Shao","year":"2024","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132076_bib0029","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2025.127417","article-title":"LEAT: Towards robust deepfake disruption in real-world scenarios via latent ensemble attack","volume":"279","author":"Shim","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132076_bib0030","series-title":"3rd International conference on learning representations (ICLR 2015)","article-title":"Very deep convolutional networks for large-scale image recognition","author":"Simonyan","year":"2015"},{"key":"10.1016\/j.eswa.2026.132076_bib0031","series-title":"Companion proceedings of the web conference 2022","first-page":"726","article-title":"Leveraging intra and inter modality relationship for multimodal fake news detection","author":"Singhal","year":"2022"},{"key":"10.1016\/j.eswa.2026.132076_bib0032","series-title":"2019 IEEE Fifth international conference on multimedia big data (bigMM)","first-page":"39","article-title":"SpotFake: A multi-modal framework for fake news detection","author":"Singhal","year":"2019"},{"key":"10.1016\/j.eswa.2026.132076_bib0033","series-title":"Proceedings of the conference. association for computational linguistics. meeting","first-page":"6558","article-title":"Multimodal transformer for unaligned multimodal language sequences","volume":"vol. 2019","author":"Tsai","year":"2019"},{"key":"10.1016\/j.eswa.2026.132076_bib0034","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2025.127741","article-title":"SEPM: Multiscale semantic enhancement-progressive multimodal fusion network for fake news detection","volume":"283","author":"Wang","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132076_bib0035","series-title":"ICASSP 2024 - 2024 IEEE International conference on acoustics, speech and signal processing (icassp)","first-page":"4935","article-title":"Exploiting modality-specific features for multi-modal manipulation detection and grounding","author":"Wang","year":"2024"},{"key":"10.1016\/j.eswa.2026.132076_bib0036","series-title":"Proceedings of the 31st ACM international conference on multimedia","first-page":"5696","article-title":"Cross-modal contrastive learning for multimodal fake news detection","author":"Wang","year":"2023"},{"key":"10.1016\/j.eswa.2026.132076_bib0037","series-title":"Proceedings of the 24th acm sigkdd international conference on knowledge discovery & data mining","first-page":"849","article-title":"EANN: Event adversarial neural networks for multi-modal fake news detection","author":"Wang","year":"2018"},{"key":"10.1016\/j.eswa.2026.132076_bib0038","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.111485","article-title":"Balanced multi-modal learning with hierarchical fusion for fake news detection","volume":"164","author":"Wu","year":"2025","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.132076_bib0039","series-title":"Proceedings of the 2020 conference on empirical methods in natural language processing (EMNLP)","first-page":"6442","article-title":"LUKE: Deep contextualized entity representations with entity-aware self-attention","author":"Yamada","year":"2020"},{"key":"10.1016\/j.eswa.2026.132076_bib0040","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110526","article-title":"Uncertainty-aware hierarchical labeling for face forgery detection","volume":"153","author":"Yu","year":"2024","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.132076_bib0041","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"10790","article-title":"Learning modality-specific representations with self-supervised multi-task learning for multimodal sentiment analysis","volume":"35","author":"Yu","year":"2021"},{"key":"10.1016\/j.eswa.2026.132076_bib0042","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1007\/s12559-024-10316-x","article-title":"Multi-modal generative deepfake detection via visual-language pretraining with gate fusion for cognitive computation","volume":"16","author":"Zhang","year":"2024","journal-title":"Cognitive Computation"},{"key":"10.1016\/j.eswa.2026.132076_bib0043","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111253","article-title":"Distilled transformers with locally enhanced global representations for face forgery detection","volume":"161","author":"Zhang","year":"2025","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.132076_bib0044","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"4005","article-title":"ASAP: Advancing semantic alignment promotes multi-modal manipulation detecting and grounding","author":"Zhang","year":"2025"},{"key":"10.1016\/j.eswa.2026.132076_bib0045","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"2185","article-title":"Multi-attentional deepfake detection","author":"Zhao","year":"2021"},{"key":"10.1016\/j.eswa.2026.132076_bib0046","series-title":"ICASSP 2024 - 2024 IEEE International conference on acoustics, speech and signal processing (ICASSP)","first-page":"8190","article-title":"Concentrated reasoning and unified reconstruction for multi-modal media manipulation","author":"Zhao","year":"2024"},{"key":"10.1016\/j.eswa.2026.132076_bib0047","unstructured":"Zhong, N., Xu, Y., Qian, Z., & Zhang, X. (2023). Rich and poor texture contrast: A simple yet effective approach for ai-generated image detection. arXiv: 2311.12397."},{"key":"10.1016\/j.eswa.2026.132076_bib0048","series-title":"Pacific-Asia conference on knowledge discovery and data mining","first-page":"354","article-title":": Similarity-aware multi-modal fake news detection","author":"Zhou","year":"2020"},{"key":"10.1016\/j.eswa.2026.132076_bib0049","series-title":"2023 IEEE International conference on multimedia and expo (ICME)","first-page":"2825","article-title":"Multimodal fake news detection via clip-guided learning","author":"Zhou","year":"2023"},{"key":"10.1016\/j.eswa.2026.132076_bib0050","series-title":"Proceedings of the 2024 joint international conference on computational linguistics, language resources and evaluation (LREC-COLING 2024)","first-page":"16581","article-title":"Towards multi-modal sarcasm detection via disentangled multi-grained multi-modal distilling","author":"Zhu","year":"2024"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426009899?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426009899?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,9]],"date-time":"2026-06-09T02:43:34Z","timestamp":1780973014000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426009899"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":50,"alternative-id":["S0957417426009899"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132076","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Multimodal feature disentangle-fusion network for detecting and grounding multi-modal media manipulation","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132076","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"132076"}}