{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T14:08:06Z","timestamp":1780495686899,"version":"3.54.1"},"reference-count":47,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100008081","name":"Southeast University","doi-asserted-by":"publisher","award":["MP202404"],"award-info":[{"award-number":["MP202404"]}],"id":[{"id":"10.13039\/501100008081","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62276061"],"award-info":[{"award-number":["62276061"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62436002"],"award-info":[{"award-number":["62436002"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.patcog.2026.113555","type":"journal-article","created":{"date-parts":[[2026,3,21]],"date-time":"2026-03-21T23:30:35Z","timestamp":1774135835000},"page":"113555","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":1,"special_numbering":"PA","title":["PLRVG: Progressive layer-wise refinement for visual grounding via deep-to-shallow decoding"],"prefix":"10.1016","volume":"179","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-5915-9565","authenticated-orcid":false,"given":"Wenxuan","family":"Cheng","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-6133-0035","authenticated-orcid":false,"given":"Ming","family":"Dai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6385-6776","authenticated-orcid":false,"given":"Wankou","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patcog.2026.113555_bib0001","series-title":"ECCV","first-page":"69","article-title":"Modeling context in referring expressions","author":"Yu","year":"2016"},{"key":"10.1016\/j.patcog.2026.113555_bib0002","series-title":"CVPR","first-page":"11","article-title":"Generation and comprehension of unambiguous object descriptions","author":"Mao","year":"2016"},{"key":"10.1016\/j.patcog.2026.113555_bib0003","series-title":"ECCV","first-page":"792","article-title":"Modeling context between objects for referring expression understanding","author":"Nagaraja","year":"2016"},{"key":"10.1016\/j.patcog.2026.113555_bib0004","article-title":"DCART: a dual contrastive alignment residual transformer model for visual grounding","volume":"172","author":"Zhu","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113555_bib0005","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.111663","article-title":"Language-Image consistency augmentation and distillation network for visual grounding","volume":"166","author":"Ke","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113555_bib0006","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.110084","article-title":"MPCCT: Multimodal vision-language learning paradigm with context-based compact transformer","volume":"147","author":"Chen","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113555_bib0007","series-title":"Advances in Neural Information Processing Systems (NeurIPS)","article-title":"SimVG: a simple framework for visual grounding with decoupled multi-modal fusion","author":"Dai","year":"2024"},{"key":"10.1016\/j.patcog.2026.113555_bib0008","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"2618","article-title":"Multi-task visual grounding with coarse-to-fine consistency constraints","volume":"Vol. 39","author":"Dai","year":"2025"},{"key":"10.1016\/j.patcog.2026.113555_bib0009","series-title":"Advances in Neural Information Processing Systems (NeurIPS)","article-title":"OneRef: unified one-tower expression grounding and segmentation with mask referring modeling","author":"Xiao","year":"2024"},{"key":"10.1016\/j.patcog.2026.113555_bib0010","series-title":"Proceedings of the 32nd ACM International Conference on Multimedia","first-page":"5460","article-title":"HiVG: hierarchical multimodal fine-grained modulation for visual grounding","author":"Xiao","year":"2024"},{"issue":"17","key":"10.1016\/j.patcog.2026.113555_bib0011","doi-asserted-by":"crossref","first-page":"2930","DOI":"10.3390\/rs17172930","article-title":"Cascaded hierarchical attention with adaptive fusion for visual grounding in remote sensing","volume":"17","author":"Zhu","year":"2025","journal-title":"Remote Sens."},{"key":"10.1016\/j.patcog.2026.113555_bib0012","doi-asserted-by":"crossref","first-page":"3979","DOI":"10.1109\/TMM.2025.3535345","article-title":"Phrase decoupling cross-modal hierarchical matching and progressive position correction for visual grounding","volume":"27","author":"Xie","year":"2024","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.patcog.2026.113555_bib0013","series-title":"European Conference on Computer Vision","first-page":"280","article-title":"Exploring plain vision transformer backbones for object detection","author":"Li","year":"2022"},{"key":"10.1016\/j.patcog.2026.113555_bib0014","series-title":"International Conference on Machine Learning (ICML)","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.patcog.2026.113555_bib0015","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition(CVPR)","article-title":"Image as a foreign language: BEiT pretraining for vision and vision-language tasks","author":"Wang","year":"2023"},{"key":"10.1016\/j.patcog.2026.113555_bib0016","series-title":"Proceedings of the International Conference on Machine Learning (ICML)","first-page":"4904","article-title":"Scaling up visual and vision-language representation learning with noisy text supervision","volume":"Vol. 139","author":"Jia","year":"2021"},{"key":"10.1016\/j.patcog.2026.113555_bib0017","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"11686","article-title":"CRIS: CLIP-driven referring image segmentation","author":"Wang","year":"2022"},{"key":"10.1016\/j.patcog.2026.113555_bib0018","doi-asserted-by":"crossref","first-page":"4334","DOI":"10.1109\/TMM.2023.3321501","article-title":"CLIP-VG: self-paced curriculum adapting of CLIP for visual grounding","volume":"26","author":"Xiao","year":"2023","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.patcog.2026.113555_bib0019","series-title":"International Conference on Machine Learning (ICML)","first-page":"5583","article-title":"ViLT: vision-and-language transformer without convolution or region supervision","author":"Kim","year":"2021"},{"key":"10.1016\/j.patcog.2026.113555_bib0020","series-title":"Advances in Neural Information Processing Systems (NeurIPS)","article-title":"Visual instruction tuning","volume":"Vol. 36","author":"Liu","year":"2024"},{"key":"10.1016\/j.patcog.2026.113555_bib0021","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"26374","article-title":"PixelLM: pixel reasoning with large multimodal model","author":"Ren","year":"2024"},{"key":"10.1016\/j.patcog.2026.113555_bib0022","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"13009","article-title":"GLaMM: pixel grounding large multimodal model","author":"Rasheed","year":"2024"},{"key":"10.1016\/j.patcog.2026.113555_bib0023","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","article-title":"MAttNet: modular attention network for referring expression comprehension","author":"Yu","year":"2018"},{"key":"10.1016\/j.patcog.2026.113555_bib0024","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111222","article-title":"Graph-based referring expression comprehension with expression-guided selective filtering and noun-oriented reasoning","volume":"161","author":"Ke","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113555_bib0025","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"13449","article-title":"ScanFormer: referring expression comprehension by iteratively scanning","author":"Su","year":"2024"},{"key":"10.1016\/j.patcog.2026.113555_bib0026","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"18155","article-title":"LAVT: language-aware vision transformer for referring image segmentation","author":"Yang","year":"2022"},{"key":"10.1016\/j.patcog.2026.113555_bib0027","series-title":"Proceedings of the European Conference on Computer Vision (ECCV)","first-page":"213","article-title":"End-to-end object detection with transformers","author":"Carion","year":"2020"},{"key":"10.1016\/j.patcog.2026.113555_bib0028","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)","first-page":"1769","article-title":"TransVG: end-to-end visual grounding with transformers","author":"Deng","year":"2021"},{"key":"10.1016\/j.patcog.2026.113555_bib0029","series-title":"European Conference on Computer Vision (ECCV)","first-page":"598","article-title":"SeqTR: a simple yet universal network for visual grounding","author":"Zhu","year":"2022"},{"key":"10.1016\/j.patcog.2026.113555_bib0030","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"9499","article-title":"Improving visual grounding with visual-linguistic verification and iterative reasoning","author":"Yang","year":"2022"},{"issue":"5","key":"10.1016\/j.patcog.2026.113555_bib0031","doi-asserted-by":"crossref","first-page":"3213","DOI":"10.1109\/TPAMI.2023.3339628","article-title":"Context disentangling and prototype inheriting for robust visual grounding","volume":"46","author":"Tang","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.113555_bib0032","first-page":"1","article-title":"Visual grounding with joint multimodal representation and interaction","volume":"72","author":"Zhu","year":"2023","journal-title":"IEEE Trans. Instrum. Meas."},{"key":"10.1016\/j.patcog.2026.113555_bib0033","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"10857","article-title":"Language adaptive weight generation for multi-task visual grounding","author":"Su","year":"2023"},{"key":"10.1016\/j.patcog.2026.113555_bib0034","article-title":"Dynamic MDETR: a dynamic multimodal transformer decoder for visual grounding","author":"Shi","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell. (TPAMI)"},{"issue":"11","key":"10.1016\/j.patcog.2026.113555_bib0035","doi-asserted-by":"crossref","first-page":"13636","DOI":"10.1109\/TPAMI.2023.3296823","article-title":"TransVG++: end-to-end visual grounding with language conditioned vision transformer","volume":"45","author":"Deng","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.113555_bib0036","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"15502","article-title":"Shifting more attention to visual backbone: query-modulated refinement networks for end-to-end visual grounding","author":"Ye","year":"2022"},{"key":"10.1016\/j.patcog.2026.113555_bib0037","series-title":"European Conference on Computer Vision","first-page":"57","article-title":"SegVG: transferring object bounding box to segmentation for visual grounding","author":"Kang","year":"2024"},{"key":"10.1016\/j.patcog.2026.113555_bib0038","series-title":"European Conference on Computer Vision (ECCV)","article-title":"UNITER: universal image-text representation learning","author":"Chen","year":"2020"},{"key":"10.1016\/j.patcog.2026.113555_bib0039","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)","first-page":"1780","article-title":"MDETR-modulated detection for end-to-end multi-modal understanding","author":"Kamath","year":"2021"},{"key":"10.1016\/j.patcog.2026.113555_bib0040","series-title":"European Conference on Computer Vision","first-page":"3","article-title":"YORO-lightweight end to end visual grounding","author":"Ho","year":"2022"},{"key":"10.1016\/j.patcog.2026.113555_bib0041","series-title":"Advances in Neural Information Processing Systems (NeurIPS)","article-title":"Referring transformer: a one-step approach to multi-task visual grounding","volume":"Vol. 34","author":"Li","year":"2021"},{"key":"10.1016\/j.patcog.2026.113555_bib0042","series-title":"European Conference on Computer Vision (ECCV)","first-page":"521","article-title":"UniTAB: unifying text and box outputs for grounded vision-language modeling","author":"Yang","year":"2022"},{"key":"10.1016\/j.patcog.2026.113555_bib0043","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence (AAAI)","first-page":"1728","article-title":"DQ-DETR: Dual query detection transformer for phrase extraction and grounding","volume":"37","author":"Liu","year":"2023"},{"key":"10.1016\/j.patcog.2026.113555_bib0044","series-title":"European Conference on Computer Vision","first-page":"38","article-title":"Grounding DINO: marrying DINO with grounded pre-training for open-set object detection","author":"Liu","year":"2024"},{"key":"10.1016\/j.patcog.2026.113555_bib0045","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"18653","article-title":"PolyFormer: referring image segmentation as sequential polygon generation","author":"Liu","year":"2023"},{"key":"10.1016\/j.patcog.2026.113555_bib0046","series-title":"International Conference on Machine Learning","first-page":"23318","article-title":"OFA: unifying architectures, tasks, and modalities through a simple sequence-to-sequence learning framework","author":"Wang","year":"2022"},{"key":"10.1016\/j.patcog.2026.113555_bib0047","series-title":"European Conference on Computer Vision","first-page":"125","article-title":"An efficient and effective transformer decoder-based framework for multi-task visual grounding","author":"Chen","year":"2024"}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326005212?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326005212?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T13:08:33Z","timestamp":1780492113000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326005212"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":47,"alternative-id":["S0031320326005212"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113555","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"PLRVG: Progressive layer-wise refinement for visual grounding via deep-to-shallow decoding","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113555","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"113555"}}