{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T14:48:40Z","timestamp":1782485320944,"version":"3.54.5"},"reference-count":53,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100011525","name":"South China University of Technology Guangdong Key Laboratory of Fermentation and Enzyme Engineering","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100011525","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1016\/j.patcog.2026.114223","type":"journal-article","created":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T06:42:09Z","timestamp":1781592129000},"page":"114223","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PC","title":["Zero-shot referring expression comprehension via guidance of Multimodal Large Language Models"],"prefix":"10.1016","volume":"180","author":[{"given":"Rouyi","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wei","family":"Zhuo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shiyi","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhihao","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1420-0815","authenticated-orcid":false,"given":"Linlin","family":"Shen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"6","key":"10.1016\/j.patcog.2026.114223_b1","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3295748","article-title":"A comprehensive survey of deep learning for image captioning","volume":"51","author":"Hossain","year":"2019","journal-title":"ACM Comput. Surv. (CsUR)"},{"issue":"1","key":"10.1016\/j.patcog.2026.114223_b2","doi-asserted-by":"crossref","first-page":"539","DOI":"10.1109\/TPAMI.2022.3148210","article-title":"From show to tell: A survey on deep learning-based image captioning","volume":"45","author":"Stefanini","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.114223_b3","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110555","article-title":"CAST: Cross-modal retrieval and visual conditioning for image captioning","volume":"153","author":"Cao","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114223_b4","doi-asserted-by":"crossref","unstructured":"K. Marino, M. Rastegari, A. Farhadi, R. Mottaghi, Ok-vqa: A visual question answering benchmark requiring external knowledge, in: Proceedings of the IEEE\/Cvf Conference on Computer Vision and Pattern Recognition, 2019, pp. 3195\u20133204.","DOI":"10.1109\/CVPR.2019.00331"},{"key":"10.1016\/j.patcog.2026.114223_b5","doi-asserted-by":"crossref","unstructured":"Z. Yu, J. Yu, Y. Cui, D. Tao, Q. Tian, Deep modular co-attention networks for visual question answering, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2019, pp. 6281\u20136290.","DOI":"10.1109\/CVPR.2019.00644"},{"key":"10.1016\/j.patcog.2026.114223_b6","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.109848","article-title":"Encoder\u2013decoder cycle for visual question answering based on perception-action cycle","volume":"144","author":"Mohamud","year":"2023","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114223_b7","doi-asserted-by":"crossref","unstructured":"J. Gu, E. Stefani, Q. Wu, J. Thomason, X. Wang, Vision-and-language navigation: A survey of tasks, methods, and future directions, in: Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics, 2022, pp. 7606\u20137623.","DOI":"10.18653\/v1\/2022.acl-long.524"},{"key":"10.1016\/j.patcog.2026.114223_b8","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110511","article-title":"Memory-adaptive vision-and-language navigation","volume":"153","author":"He","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114223_b9","series-title":"Adapting clip for phrase localization without further training","author":"Li","year":"2022"},{"key":"10.1016\/j.patcog.2026.114223_b10","series-title":"ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"2765","article-title":"Vgdiffzero: Text-to-image diffusion models can be zero-shot visual grounders","author":"Liu","year":"2024"},{"key":"10.1016\/j.patcog.2026.114223_b11","series-title":"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), ACL 2022, Dublin, Ireland, May 22-27, 2022","first-page":"5198","article-title":"ReCLIP: A strong zero-shot baseline for referring expression comprehension","author":"Subramanian","year":"2022"},{"issue":"7","key":"10.1016\/j.patcog.2026.114223_b12","first-page":"7487","article-title":"Rethinking two-stage referring expression comprehension: A novel grounding and segmentation method modulated by point","volume":"38","author":"Zhao","year":"2024","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"issue":"2","key":"10.1016\/j.patcog.2026.114223_b13","doi-asserted-by":"crossref","DOI":"10.1145\/3777449","article-title":"Implement referring expression comprehension by extending auto-focus lens to locked vision model","volume":"22","author":"Zheng","year":"2026","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."},{"key":"10.1016\/j.patcog.2026.114223_b14","doi-asserted-by":"crossref","first-page":"4426","DOI":"10.1109\/TMM.2020.3042066","article-title":"Referring expression comprehension: A survey of methods and datasets","volume":"23","author":"Qiao","year":"2020","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.patcog.2026.114223_b15","doi-asserted-by":"crossref","unstructured":"J. Mao, J. Huang, A. Toshev, O. Camburu, A.L. Yuille, K. Murphy, Generation and comprehension of unambiguous object descriptions, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016, pp. 11\u201320.","DOI":"10.1109\/CVPR.2016.9"},{"key":"10.1016\/j.patcog.2026.114223_b16","doi-asserted-by":"crossref","unstructured":"S. Yu, P.H. Seo, J. Son, Zero-shot referring image segmentation with global-local context features, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 19456\u201319465.","DOI":"10.1109\/CVPR52729.2023.01864"},{"key":"10.1016\/j.patcog.2026.114223_b17","series-title":"International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.patcog.2026.114223_b18","first-page":"9694","article-title":"Align before fuse: Vision and language representation learning with momentum distillation","volume":"34","author":"Li","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114223_b19","doi-asserted-by":"crossref","unstructured":"L. Yu, Z. Lin, X. Shen, J. Yang, X. Lu, M. Bansal, T.L. Berg, Mattnet: Modular attention network for referring expression comprehension, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018, pp. 1307\u20131315.","DOI":"10.1109\/CVPR.2018.00142"},{"key":"10.1016\/j.patcog.2026.114223_b20","series-title":"European Conference on Computer Vision","first-page":"350","article-title":"Detecting twenty-thousand classes using image-level supervision","author":"Zhou","year":"2022"},{"issue":"11","key":"10.1016\/j.patcog.2026.114223_b21","doi-asserted-by":"crossref","first-page":"4189","DOI":"10.1109\/TPAMI.2021.3058684","article-title":"Discriminative triad matching and reconstruction for weakly referring expression grounding","volume":"43","author":"Sun","year":"2021","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.114223_b22","doi-asserted-by":"crossref","unstructured":"J. Chen, W. Shen, Z. Wei, L. Sun, H. Zhang, Leveraging Debiased Cross-modal Attention Maps and Code-based Reasoning for Zero-shot Referring Expression Comprehension, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2025, pp. 20413\u201320424.","DOI":"10.1109\/ICCV51701.2025.01898"},{"key":"10.1016\/j.patcog.2026.114223_b23","doi-asserted-by":"crossref","unstructured":"Z. Han, F. Zhu, Q. Lao, H. Jiang, Zero-shot referring expression comprehension via structural similarity between images and captions, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 14364\u201314374.","DOI":"10.1109\/CVPR52733.2024.01362"},{"key":"10.1016\/j.patcog.2026.114223_b24","doi-asserted-by":"crossref","unstructured":"J. Chen, F. Wei, J. Zhao, S. Song, B. Wu, Z. Peng, S.-H.G. Chan, H. Zhang, Revisiting referring expression comprehension evaluation in the era of large multimodal models, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2025, pp. 513\u2013524.","DOI":"10.1109\/CVPRW67362.2025.00056"},{"key":"10.1016\/j.patcog.2026.114223_b25","article-title":"Improving scene knowledge referring expression comprehension with large language models","author":"Li","year":"2025","journal-title":"IEEE MultiMedia"},{"key":"10.1016\/j.patcog.2026.114223_b26","doi-asserted-by":"crossref","unstructured":"G. Li, M. Zhang, X. Zheng, P. Chen, Z. Wang, Y. Shen, M. Zhuge, C. Wu, F. Chao, K. Li, et al., Multimodal Inplace Prompt Tuning for Open-set Object Detection, in: Proceedings of the 32nd ACM International Conference on Multimedia, 2024, pp. 8062\u20138071.","DOI":"10.1145\/3664647.3681275"},{"issue":"11","key":"10.1016\/j.patcog.2026.114223_b27","doi-asserted-by":"crossref","first-page":"16277","DOI":"10.1109\/TNNLS.2023.3293484","article-title":"Fine-grained visual\u2013text prompt-driven self-training for open-vocabulary object detection","volume":"35","author":"Long","year":"2023","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"10.1016\/j.patcog.2026.114223_b28","doi-asserted-by":"crossref","unstructured":"J. Qin, J. Wu, P. Yan, M. Li, R. Yuxi, X. Xiao, Y. Wang, R. Wang, S. Wen, X. Pan, et al., Freeseg: Unified, universal and open-vocabulary image segmentation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 19446\u201319455.","DOI":"10.1109\/CVPR52729.2023.01863"},{"key":"10.1016\/j.patcog.2026.114223_b29","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.111663","article-title":"Language\u2013Image consistency augmentation and distillation network for visual grounding","volume":"166","author":"Ke","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114223_b30","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.110084","article-title":"MPCCT: Multimodal vision-language learning paradigm with context-based compact transformer","volume":"147","author":"Chen","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114223_b31","doi-asserted-by":"crossref","unstructured":"T. Gupta, A. Kembhavi, Visual programming: Compositional visual reasoning without training, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 14953\u201314962.","DOI":"10.1109\/CVPR52729.2023.01436"},{"issue":"3","key":"10.1016\/j.patcog.2026.114223_b32","article-title":"Structure-CLIP: Towards scene graph knowledge to enhance multi-modal structured representations.","volume":"38","author":"Huang","year":"2024","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"10.1016\/j.patcog.2026.114223_b33","doi-asserted-by":"crossref","unstructured":"K. Jiang, X. He, R. Xu, X.E. Wang, ComCLIP: Training-Free Compositional Image and Text Matching., in: Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics, 2022, pp. 6639\u20136659.","DOI":"10.18653\/v1\/2024.naacl-long.370"},{"key":"10.1016\/j.patcog.2026.114223_b34","doi-asserted-by":"crossref","unstructured":"J. Liu, L. Wang, M.-H. Yang, Referring expression generation and comprehension via attributes, in: Proceedings of the IEEE International Conference on Computer Vision, 2017, pp. 4856\u20134864.","DOI":"10.1109\/ICCV.2017.520"},{"key":"10.1016\/j.patcog.2026.114223_b35","doi-asserted-by":"crossref","unstructured":"S. Yang, G. Li, Y. Yu, Dynamic graph attention for referring expression comprehension, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2019, pp. 4644\u20134653.","DOI":"10.1109\/ICCV.2019.00474"},{"key":"10.1016\/j.patcog.2026.114223_b36","doi-asserted-by":"crossref","first-page":"30","DOI":"10.1016\/j.aiopen.2024.01.004","article-title":"Cpt: Colorful prompt tuning for pre-trained vision-language models","volume":"5","author":"Yao","year":"2024","journal-title":"AI Open"},{"key":"10.1016\/j.patcog.2026.114223_b37","doi-asserted-by":"crossref","unstructured":"R. Girshick, Fast r-cnn, in: Proceedings of the IEEE International Conference on Computer Vision, 2015, pp. 1440\u20131448.","DOI":"10.1109\/ICCV.2015.169"},{"issue":"2","key":"10.1016\/j.patcog.2026.114223_b38","first-page":"1656","article-title":"Look around before locating: Considering content and structure information for visual grounding","volume":"39","author":"Zheng","year":"2025","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"10.1016\/j.patcog.2026.114223_b39","doi-asserted-by":"crossref","unstructured":"H. Shen, T. Zhao, M. Zhu, J. Yin, Groundvlp: Harnessing zero-shot visual grounding from vision-language pre-training and open-vocabulary object detection, in: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, 2024, pp. 4766\u20134775.","DOI":"10.1609\/aaai.v38i5.28278"},{"key":"10.1016\/j.patcog.2026.114223_b40","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110779","article-title":"Few-shot relational triple extraction with hierarchical prototype optimization","volume":"156","author":"Gao","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114223_b41","doi-asserted-by":"crossref","unstructured":"R.R. Selvaraju, M. Cogswell, A. Das, R. Vedantam, D. Parikh, D. Batra, Grad-cam: Visual explanations from deep networks via gradient-based localization, in: Proceedings of the IEEE International Conference on Computer Vision, 2017, pp. 618\u2013626.","DOI":"10.1109\/ICCV.2017.74"},{"key":"10.1016\/j.patcog.2026.114223_b42","series-title":"VLMAE: Vision-language masked autoencoder","author":"He","year":"2022"},{"key":"10.1016\/j.patcog.2026.114223_b43","series-title":"Semantic abstraction: Open-world 3d scene understanding from 2d vision-language models","author":"Ha","year":"2022"},{"key":"10.1016\/j.patcog.2026.114223_b44","doi-asserted-by":"crossref","first-page":"23716","DOI":"10.52202\/068431-1723","article-title":"Flamingo: A visual language model for few-shot learning","volume":"35","author":"Alayrac","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114223_b45","article-title":"Carbongpt: meta causal graph-enhanced large language models for carbon emission forecasting in large-scale power distribution networks","volume":"early access","author":"Li","year":"2026","journal-title":"IEEE Trans. Smart Grid"},{"issue":"8","key":"10.1016\/j.patcog.2026.114223_b46","doi-asserted-by":"crossref","first-page":"3825","DOI":"10.1109\/TCYB.2025.3569333","article-title":"Causal intervention is what large language models need for spatio-temporal forecasting","volume":"55","author":"Li","year":"2025","journal-title":"IEEE Trans. Cybern."},{"key":"10.1016\/j.patcog.2026.114223_b47","doi-asserted-by":"crossref","DOI":"10.1016\/j.apenergy.2025.127227","article-title":"Rsynllm: a risk-aware routing mixture-of-experts large language model for multi-energy load forecasting in large-scale distribution networks","volume":"406","author":"Li","year":"2026","journal-title":"Appl. Energy"},{"issue":"8","key":"10.1016\/j.patcog.2026.114223_b48","doi-asserted-by":"crossref","first-page":"2765","DOI":"10.1109\/TPAMI.2020.2973983","article-title":"Relationship-embedded representation learning for grounding referring expressions","volume":"43","author":"Yang","year":"2020","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.114223_b49","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"6","key":"10.1016\/j.patcog.2026.114223_b50","doi-asserted-by":"crossref","first-page":"5500","DOI":"10.1109\/TSG.2024.3408640","article-title":"Real-time robust state estimation for large-scale low-observability power-transportation system based on meta physics-informed graph timesnet","volume":"15","author":"Li","year":"2024","journal-title":"IEEE Trans. Smart Grid"},{"key":"10.1016\/j.patcog.2026.114223_b51","series-title":"Computer Vision\u2013ECCV 2016: 14th European Conference, Amsterdam, the Netherlands, October 11-14, 2016, Proceedings, Part II 14","first-page":"69","article-title":"Modeling context in referring expressions","author":"Yu","year":"2016"},{"key":"10.1016\/j.patcog.2026.114223_b52","series-title":"Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13","first-page":"740","article-title":"Microsoft coco: Common objects in context","author":"Lin","year":"2014"},{"key":"10.1016\/j.patcog.2026.114223_b53","unstructured":"L.H. Li, P. Zhang, H. Zhang, J. Yang, C. Li, Y. Zhong, L. Wang, L. Yuan, L. Zhang, J.-N. Hwang, et al., Grounded language-image pre-training, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 10965\u201310975."}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S003132032601188X?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S003132032601188X?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T14:06:42Z","timestamp":1782482802000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S003132032601188X"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,12]]},"references-count":53,"alternative-id":["S003132032601188X"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.114223","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Zero-shot referring expression comprehension via guidance of Multimodal Large Language Models","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.114223","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"114223"}}