{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,8]],"date-time":"2026-03-08T18:42:33Z","timestamp":1772995353797,"version":"3.50.1"},"reference-count":62,"publisher":"Institution of Engineering and Technology (IET)","issue":"1","license":[{"start":{"date-parts":[[2026,2,19]],"date-time":"2026-02-19T00:00:00Z","timestamp":1771459200000},"content-version":"vor","delay-in-days":49,"URL":"http:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/doi.wiley.com\/10.1002\/tdm_license_1.1"}],"content-domain":{"domain":["ietresearch.onlinelibrary.wiley.com"],"crossmark-restriction":true},"short-container-title":["IET Image Processing"],"published-print":{"date-parts":[[2026,1]]},"abstract":"<jats:title>ABSTRACT<\/jats:title>\n                  <jats:p>Accurate polyp segmentation from a single color image remains a significant challenge due to the complex appearance of lesions and the lack of diverse contextual priors. Existing methods usually rely on a limited image prior, which leads to suboptimal results. We propose a novel approach that leverages rich contextual information from multiple modalities (including depth, higher order semantics, and edge features) to produce precise segmentation results. By integrating the Depth Anything model, Large Language Models, and Edge Extractor, our method effectively fuses these diverse priors to overcome the limitations of single modality approaches. In addition, we introduce a diffusion modeling framework to bring powerful generative priors for endoscopic image segmentation. This is a flexible deep network architecture that efficiently fuses multimodal information and can accommodate any number of multimodal inputs. Extensive experimental results demonstrate that our approach achieves promising performance on several\u00a0benchmarks.<\/jats:p>","DOI":"10.1049\/ipr2.70305","type":"journal-article","created":{"date-parts":[[2026,2,19]],"date-time":"2026-02-19T09:32:47Z","timestamp":1771493567000},"update-policy":"https:\/\/doi.org\/10.1002\/crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["The Power of Modality: Improving Polyp Segmentation With Multimodal Information"],"prefix":"10.1049","volume":"20","author":[{"given":"Fang","family":"Wang","sequence":"first","affiliation":[{"name":"School of Mathematics and Statistics Zaozhuang University Zaozhuang China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pu","family":"Wang","sequence":"additional","affiliation":[{"name":"Hetao College Shenzhen China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Meng","family":"Zhao","sequence":"additional","affiliation":[{"name":"School of Information Science and Engineering Zaozhuang University Zaozhuang China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chenggang","family":"Shan","sequence":"additional","affiliation":[{"name":"School of Information Science and Engineering Zaozhuang University Zaozhuang China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-9717-2778","authenticated-orcid":false,"given":"Zhen","family":"Yang","sequence":"additional","affiliation":[{"name":"School of Information Science and Engineering Zaozhuang University Zaozhuang China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"265","published-online":{"date-parts":[[2026,2,19]]},"reference":[{"key":"e_1_2_12_2_1","doi-asserted-by":"publisher","DOI":"10.1002\/cncr.33587"},{"key":"e_1_2_12_3_1","doi-asserted-by":"publisher","DOI":"10.3390\/cancers13092025"},{"key":"e_1_2_12_4_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.imavis.2025.105648"},{"key":"e_1_2_12_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2021.3063716"},{"key":"e_1_2_12_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIM.2023.3244219"},{"key":"e_1_2_12_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-023-10621-1"},{"key":"e_1_2_12_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2025.3591721"},{"key":"e_1_2_12_9_1","doi-asserted-by":"publisher","DOI":"10.1148\/rg.2018170110"},{"key":"e_1_2_12_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00050"},{"key":"e_1_2_12_11_1","doi-asserted-by":"publisher","DOI":"10.1002\/ima.22568"},{"key":"e_1_2_12_12_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41598-023-36940-5"},{"key":"e_1_2_12_13_1","unstructured":"P.Wang H.Ma Z.Zhang andZ.Zheng \u201cPolypFlow: Reinforcing Polyp Segmentation with Flow\u2010Driven Dynamics \u201darXiv preprint arXiv:2502.19037 (2025)."},{"key":"e_1_2_12_14_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-87193-2_66"},{"key":"e_1_2_12_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISM46123.2019.00049"},{"key":"e_1_2_12_16_1","doi-asserted-by":"crossref","unstructured":"Z.Zheng P.Wang W.Liu J.Li R.Ye andD.Ren \u201cDistance\u2010IOU Loss: Faster and Better Learning for Bounding Box Regression \u201d inProceedings of the AAAI Conference on Artificial Intelligence vol.34(AAAI Press 2020) 12993\u201313000.","DOI":"10.1609\/aaai.v34i07.6999"},{"key":"e_1_2_12_17_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jacc.2006.04.026"},{"key":"e_1_2_12_18_1","doi-asserted-by":"publisher","DOI":"10.1186\/s12967-025-06428-z"},{"key":"e_1_2_12_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02520"},{"key":"e_1_2_12_20_1","unstructured":"W.Hong W.Wang M.Ding et\u00a0al. \u201cCogVLM2: Visual Language Models for Image and Video Understanding \u201darXiv preprint arXiv:2408.16500(2024)."},{"key":"e_1_2_12_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3369699"},{"key":"e_1_2_12_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01354"},{"key":"e_1_2_12_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00695"},{"key":"e_1_2_12_24_1","doi-asserted-by":"publisher","DOI":"10.1155\/2021\/5538927"},{"key":"e_1_2_12_25_1","doi-asserted-by":"publisher","DOI":"10.1016\/S1076-6332(03)00671-8"},{"key":"e_1_2_12_26_1","unstructured":"Y.JiandS.Gao \u201cEvaluating the Effectiveness of Large Language Models in Representing Textual Descriptions of Geometry and Spatial Relations \u201darXiv preprint arXiv:2307.03678 (2023)."},{"key":"e_1_2_12_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01046"},{"key":"e_1_2_12_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3560815"},{"key":"e_1_2_12_29_1","first-page":"51008","article-title":"Hard Prompts Made Easy: Gradient\u2010Based Discrete Optimization for Prompt Tuning and Discovery","volume":"36","author":"Wen Y.","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_2_12_30_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.bspc.2023.105934"},{"key":"e_1_2_12_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/TITB.2003.813794"},{"key":"e_1_2_12_32_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0031-3203(98)00038-7"},{"key":"e_1_2_12_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2004.826758"},{"key":"e_1_2_12_34_1","unstructured":"B.Kayalibay G.Jensen andP.Van Der Smagt \u201cCNN\u2010Based Segmentation of Medical Imaging Data \u201darXiv preprint arXiv:1701.03056 (2017)."},{"key":"e_1_2_12_35_1","doi-asserted-by":"crossref","unstructured":"O.Ronneberger P.Fischer andT.Brox \u201cU\u2010Net: Convolutional Networks for Biomedical Image Segmentation \u201d inMICCAI(Springer 2015).","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"e_1_2_12_36_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.isprsjprs.2020.01.013"},{"key":"e_1_2_12_37_1","doi-asserted-by":"crossref","unstructured":"Z.Zhou M. M. R.Siddiquee N.Tajbakhsh andJ.Liang \u201cUNet++: A Nested U\u2010Net Architecture for Medical Image Segmentation \u201d inMICCAI(Springer 2018).","DOI":"10.1007\/978-3-030-00889-5_1"},{"issue":"4","key":"e_1_2_12_38_1","first-page":"24","article-title":"Attention\u2010UNet: A Deep Learning Approach for Fast and Accurate Segmentation in Medical Imaging","volume":"2","author":"Zhu Z.","year":"2022","journal-title":"Journal of Computer Science and Software Applications"},{"key":"e_1_2_12_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00326"},{"key":"e_1_2_12_40_1","doi-asserted-by":"crossref","unstructured":"D.\u2010P.Fan G.\u2010P.Ji T.Zhou et\u00a0al. \u201cPRANet: Parallel Reverse Attention Network for Polyp Segmentation \u201d inInternational Conference on Medical Image Computing and Computer\u2010Assisted Intervention(Springer 2020) 263\u2013273.","DOI":"10.1007\/978-3-030-59725-2_26"},{"key":"e_1_2_12_41_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.media.2024.103280"},{"key":"e_1_2_12_42_1","first-page":"205","volume-title":"European Conference on omputer vision","author":"Cao H.","year":"2022"},{"key":"e_1_2_12_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/ASPCON49795.2020.9276732"},{"key":"e_1_2_12_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISBI52829.2022.9761447"},{"key":"e_1_2_12_45_1","doi-asserted-by":"crossref","unstructured":"J.Jamrasnarodom P.Rajborirug P.Pisespongsa andK.Pasupa \u201cOptimizing Colorectal Polyp Detection and Localization: Impact of RGB Color Adjustment on CNN Performance \u201dMethodsX(2025):103187.","DOI":"10.1016\/j.mex.2025.103187"},{"key":"e_1_2_12_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2020.2968250"},{"key":"e_1_2_12_47_1","first-page":"1","volume-title":"2025 International Conference on Wireless Communications Signal Processing and Networking (WiSPNET)","author":"Sushma B.","year":"2025"},{"key":"e_1_2_12_48_1","doi-asserted-by":"crossref","unstructured":"B.Li D.Zhang Z.Zhao J.Gao andX.Li \u201cU3m: Unbiased Multiscale Modal Fusion Model for Multimodal Semantic Segmentation \u201darXiv preprint arXiv:2405.15365 (2024).","DOI":"10.1016\/j.patcog.2025.111801"},{"key":"e_1_2_12_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2019.2919139"},{"key":"e_1_2_12_50_1","unstructured":"Z.HuandD.Xu \u201cVideocontrolnet: A Motion\u2010Guided Video\u2010to\u2010Video Translation Framework by Using Diffusion Model with ControlNet \u201darXiv preprint arXiv:2307.14073 (2023)."},{"key":"e_1_2_12_51_1","unstructured":"H.Ye J.Zhang S.Liu X.Han andW.Yang \u201cIP\u2010Adapter: Text Compatible Image Prompt Adapter for Text\u2010to\u2010Image Diffusion Models \u201darXiv preprint arXiv:2308.06721 (2023)."},{"key":"e_1_2_12_52_1","unstructured":"J.Yu X.Li J. Y.Koh et\u00a0al. \u201cVector\u2010Quantized Image Modeling with Improved VQGan \u201darXiv preprint arXiv:2110.04627 (2021)."},{"key":"e_1_2_12_53_1","unstructured":"A.Nichol P.Dhariwal A.Ramesh et\u00a0al. \u201cGlide: Towards Photorealistic Image Generation and Editing with Text\u2010Guided Diffusion Models \u201darXiv preprintarXiv:2112.10741 (2021)."},{"key":"e_1_2_12_54_1","unstructured":"S.Narzary B.Brahma H.Mahilary M.Brahma B.Som andS.Nandi \u201cComparative Study of Zero\u2010Shot Cross\u2010Lingual Transfer for BODO POS and NER Tagging Using Gemini 2.0 Flash Thinking Experimental Model \u201darXiv preprint arXiv:2503.04405 (2025)."},{"key":"e_1_2_12_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00987"},{"key":"e_1_2_12_56_1","first-page":"6585","volume-title":"ICASSP 2024\u20102024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","author":"Shagidanov A.","year":"2024"},{"key":"e_1_2_12_57_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-37734-2_37"},{"key":"e_1_2_12_58_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.compmedimag.2015.02.007"},{"issue":"2","key":"e_1_2_12_59_1","first-page":"630","article-title":"Automated Polyp Detection in Colonoscopy Videos Using Shape and Context Information","volume":"35","author":"Tajbakhsh N.","year":"2015","journal-title":"IEEE TMI"},{"key":"e_1_2_12_60_1","doi-asserted-by":"publisher","DOI":"10.1155\/2017\/4037190"},{"key":"e_1_2_12_61_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11548-013-0926-3"},{"key":"e_1_2_12_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2024.3461654"},{"key":"e_1_2_12_63_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2023.121754"}],"container-title":["IET Image Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/ietresearch.onlinelibrary.wiley.com\/doi\/pdf\/10.1049\/ipr2.70305","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/ietresearch.onlinelibrary.wiley.com\/doi\/full-xml\/10.1049\/ipr2.70305","content-type":"application\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/ietresearch.onlinelibrary.wiley.com\/doi\/pdf\/10.1049\/ipr2.70305","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,8]],"date-time":"2026-03-08T15:37:31Z","timestamp":1772984251000},"score":1,"resource":{"primary":{"URL":"https:\/\/ietresearch.onlinelibrary.wiley.com\/doi\/10.1049\/ipr2.70305"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1]]},"references-count":62,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2026,1]]}},"alternative-id":["10.1049\/ipr2.70305"],"URL":"https:\/\/doi.org\/10.1049\/ipr2.70305","archive":["Portico"],"relation":{},"ISSN":["1751-9659","1751-9667"],"issn-type":[{"value":"1751-9659","type":"print"},{"value":"1751-9667","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,1]]},"assertion":[{"value":"2025-08-25","order":0,"name":"received","label":"Received","group":{"name":"publication_history","label":"Publication History"}},{"value":"2026-02-04","order":2,"name":"accepted","label":"Accepted","group":{"name":"publication_history","label":"Publication History"}},{"value":"2026-02-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}],"article-number":"e70305"}}