{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,14]],"date-time":"2025-10-14T04:11:29Z","timestamp":1760415089649,"version":"build-2065373602"},"reference-count":95,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2025,10,9]],"date-time":"2025-10-09T00:00:00Z","timestamp":1759968000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,10,9]],"date-time":"2025-10-09T00:00:00Z","timestamp":1759968000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Sci. China Inf. Sci."],"published-print":{"date-parts":[[2026,2]]},"DOI":"10.1007\/s11432-024-4416-8","type":"journal-article","created":{"date-parts":[[2025,10,14]],"date-time":"2025-10-14T01:27:23Z","timestamp":1760405243000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["PGPL: enhancing spatial awareness abilities of multimodal large language models based on precise geometric position learning"],"prefix":"10.1007","volume":"69","author":[{"given":"Yongqiang","family":"Zhao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhenyu","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhi","family":"Jin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Feng","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ziliang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lianwei","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chengfeng","family":"Dou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Haiyan","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xinhai","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,10,9]]},"reference":[{"key":"4416_CR1","doi-asserted-by":"publisher","first-page":"447","DOI":"10.1007\/s11633-022-1410-8","volume":"20","author":"X Wang","year":"2023","unstructured":"Wang X, Chen G, Qian G, et al. Large-scale multi-modal pre-trained models: a comprehensive survey. Mach Intell Res, 2023, 20: 447\u2013482","journal-title":"Mach Intell Res"},{"key":"4416_CR2","unstructured":"Liu H, Li C, Wu Q, et al. Visual instruction tuning. 2023. ArXiv:2304.08485, 2023"},{"key":"4416_CR3","unstructured":"Li J, Li D, Savarese S, et al. Blip-2: bootstrapping language-image pre-training with frozen image encoders and large language models. 2023. ArXiv:2301.12597"},{"key":"4416_CR4","unstructured":"Zhang Z, Zhang A, Li M, et al. Multimodal chain-of-thought reasoning in language models. 2023. ArXiv:2302.00923"},{"key":"4416_CR5","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"W Wang","year":"2024","unstructured":"Wang W, Chen Z, Chen X, et al. VisionLLM: large language model is also an open-ended decoder for vision-centric tasks. In: Proceedings of the Advances in Neural Information Processing Systems, 2024"},{"key":"4416_CR6","doi-asserted-by":"publisher","first-page":"102274","DOI":"10.1016\/j.lindif.2023.102274","volume":"103","author":"E Kasneci","year":"2023","unstructured":"Kasneci E, Sessler K, K\u00fcchemann S, et al. ChatGPT for good? On opportunities and challenges of large language models for education. Learn Individ Differ, 2023, 103: 102274","journal-title":"Learn Individ Differ"},{"key":"4416_CR7","doi-asserted-by":"publisher","first-page":"100017","DOI":"10.1016\/j.metrad.2023.100017","volume":"1","author":"Y Liu","year":"2023","unstructured":"Liu Y, Han T, Ma S, et al. Summary of ChatGPT-related research and perspective towards the future of large language models. Meta-Radiol, 2023, 1: 100017","journal-title":"Meta-Radiol"},{"key":"4416_CR8","volume-title":"Alpaca: a strong, replicable instruction-following model","author":"R Taori","year":"2023","unstructured":"Taori R, Gulrajani I, Zhang T, et al. Alpaca: a strong, replicable instruction-following model. 2023. https:\/\/crfm.stanford.edu\/2023\/03\/13\/alpaca.html"},{"key":"4416_CR9","volume-title":"Vicuna: an open-source chatbot impressing GPT-4 with 90% ChatGPT quality","author":"W L Chiang","year":"2023","unstructured":"Chiang W L, Li Z, Lin Z, et al. Vicuna: an open-source chatbot impressing GPT-4 with 90% ChatGPT quality. 2023. https:\/\/lmsys.org\/blog\/2023-03-30-vicuna\/"},{"key":"4416_CR10","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"S Bae","year":"2024","unstructured":"Bae S, Kyung D, Ryu J, et al. EHRXQA: a multi-modal question answering dataset for electronic health records with chest X-ray images. In: Proceedings of the Advances in Neural Information Processing Systems, 2024"},{"key":"4416_CR11","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"W Lin","year":"2024","unstructured":"Lin W, Chen J, Mei J, et al. Fine-grained late-interaction multi-modal retrieval for retrieval augmented visual question answering. In: Proceedings of the Advances in Neural Information Processing Systems, 2024"},{"key":"4416_CR12","unstructured":"Yin S, Fu C, Zhao S, et al. A survey on multimodal large language models. 2023. ArXiv:2306.13549"},{"key":"4416_CR13","doi-asserted-by":"publisher","first-page":"2938","DOI":"10.1109\/TASLP.2023.3302238","volume":"32","author":"Y Chang","year":"2024","unstructured":"Chang Y, Ko Y. Two-step masked language model for domain-adapting multi-modal task-oriented dialogue systems. IEEE ACM Trans Audio Speech Lang Process, 2024, 32: 2938\u20132943","journal-title":"IEEE ACM Trans Audio Speech Lang Process"},{"key":"4416_CR14","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"A Majumdar","year":"2024","unstructured":"Majumdar A, Yadav K, Arnaud S, et al. Where are we in the search for an artificial visual cortex for embodied intelligence? In: Proceedings of the Advances in Neural Information Processing Systems, 2024"},{"key":"4416_CR15","unstructured":"Wang J, Wu Z, Li Y, et al. Large language models for robotics: opportunities, challenges, and perspectives. 2024. ArXiv:2401.04334"},{"key":"4416_CR16","first-page":"4171","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","author":"O Ma\u00f1as","year":"2024","unstructured":"Ma\u00f1as O, Krojer B, Agrawal A. Improving automatic VQA evaluation using large language models. In: Proceedings of the AAAI Conference on Artificial Intelligence, 2024. 4171\u20134179"},{"key":"4416_CR17","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"Z Khan","year":"2024","unstructured":"Khan Z, BG V K, Schulter S, et al. Exploring question decomposition for zero-shot VQA. In: Proceedings of the Advances in Neural Information Processing Systems, 2024"},{"key":"4416_CR18","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"J Gao","year":"2024","unstructured":"Gao J, Wu Q, Blair A, et al. LoRA: a logical reasoning augmented dataset for visual question answering. In: Proceedings of the Advances in Neural Information Processing Systems, 2024"},{"key":"4416_CR19","first-page":"4542","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","author":"T Qian","year":"2024","unstructured":"Qian T, Chen J, Zhuo L, et al. Nuscenes-QA: a multi-modal visual question answering benchmark for autonomous driving scenario. In: Proceedings of the AAAI Conference on Artificial Intelligence, 2024. 4542\u20134550"},{"key":"4416_CR20","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"X Tian","year":"2024","unstructured":"Tian X, Jiang T, Yun L, et al. OCC3D: a large-scale 3D occupancy prediction benchmark for autonomous driving. In: Proceedings of the Advances in Neural Information Processing Systems, 2024"},{"key":"4416_CR21","doi-asserted-by":"crossref","unstructured":"Chen Q, Ye H, Hong Y. Med3dinsight: enhancing 3D medical image understanding with 2D multi-modal large language models. 2024. ArXiv:2403.05141","DOI":"10.1109\/JBHI.2025.3609739"},{"key":"4416_CR22","doi-asserted-by":"publisher","first-page":"355","DOI":"10.1016\/j.inffus.2021.06.007","volume":"76","author":"G Muhammad","year":"2021","unstructured":"Muhammad G, Alshehri F, Karray F, et al. A comprehensive survey on multimodal medical signals fusion for smart healthcare systems. Inf Fusion, 2021, 76: 355\u2013375","journal-title":"Inf Fusion"},{"key":"4416_CR23","unstructured":"Zhu D, Chen J, Shen X, et al. MiniGPT-4: enhancing vision-language understanding with advanced large language models. 2023. ArXiv:2304.10592"},{"key":"4416_CR24","volume-title":"Proceedings of the Conference on Empirical Methods in Natural Language Processing","author":"A Kamath","year":"2023","unstructured":"Kamath A, Hessel J, Chang K W. What\u2019s \u201cup\u201d with vision-language models? Investigating their struggle with spatial reasoning. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing, 2023"},{"key":"4416_CR25","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"Z Yin","year":"2024","unstructured":"Yin Z, Wang J, Cao J, et al. LAMM: language-assisted multi-modal instruction-tuning dataset, framework, and benchmark. In: Proceedings of the Advances in Neural Information Processing Systems, 2024"},{"key":"4416_CR26","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"A Zhang","year":"2024","unstructured":"Zhang A, Fei H, Yao Y, et al. VPGTrans: transfer visual prompt generator across LLMs. In: Proceedings of the Advances in Neural Information Processing Systems, 2024"},{"key":"4416_CR27","unstructured":"Dong Q, Li L, Dai D, et al. A survey for in-context learning. 2022. ArXiv:2301.00234"},{"key":"4416_CR28","unstructured":"Lu P, Peng B, Cheng H, et al. Chameleon: plug-and-play compositional reasoning with large language models. 2023. ArXiv:2304.09842"},{"key":"4416_CR29","doi-asserted-by":"crossref","unstructured":"Wang Q C, Xiao Z, Mao Y, et al. Model predictive task sampling for efficient and robust adaptation. 2025. ArXiv:2501.11039","DOI":"10.21203\/rs.3.rs-6700167\/v1"},{"key":"4416_CR30","unstructured":"Ye Q, Xu H, Xu G, et al. mPLUG-Owl: modularization empowers large language models with multimodality. 2023. ArXiv:2304.14178"},{"key":"4416_CR31","unstructured":"Zhang S, Sun P, Chen S, et al. GPT4ROI: instruction tuning large language model on region-of-interest. 2023. ArXiv:2307.03601"},{"key":"4416_CR32","unstructured":"Peng Z, Wang W, Dong L, et al. Kosmos-2: grounding multimodal large language models to the world. 2023. ArXiv:2306.14824"},{"key":"4416_CR33","unstructured":"Shen Y, Song K, Tan X, et al. HuggingGPT: solving AI tasks with ChatGPT and its friends in huggingface. 2023. ArXiv:2303.17580"},{"key":"4416_CR34","unstructured":"Zhao Y, Li Z, Zhang F, et al. Enhancing subtask performance of multi-modal large language model. 2023. ArXiv:2308.16474"},{"key":"4416_CR35","unstructured":"Wu C, Yin S, Qi W, et al. Visual ChatGPT: talking, drawing and editing with visual foundation models. 2023. ArXiv:2303.04671"},{"key":"4416_CR36","unstructured":"Chen K, Zhang Z, Zeng W, et al. Shikra: unleashing multimodal LLM\u2019s referential dialogue magic. 2023. ArXiv:2306.15195"},{"key":"4416_CR37","unstructured":"Zhang W, Lin T, Liu J, et al. HyperLLaVA: dynamic visual and language expert tuning for multimodal large language models. 2024. ArXiv:2403.13447"},{"key":"4416_CR38","doi-asserted-by":"publisher","first-page":"257","DOI":"10.1109\/JPROC.2023.3238524","volume":"111","author":"Z Zou","year":"2023","unstructured":"Zou Z, Chen K, Shi Z, et al. Object detection in 20 years: a survey. Proc IEEE, 2023, 111: 257\u2013276","journal-title":"Proc IEEE"},{"key":"4416_CR39","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"Y Pu","year":"2024","unstructured":"Pu Y, Liang W, Hao Y, et al. Rank-DETR for high quality object detection. In: Proceedings of the Advances in Neural Information Processing Systems, 2024"},{"key":"4416_CR40","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"L Meng","year":"2024","unstructured":"Meng L, Dai X, Yang J, et al. Learning from rich semantics and coarse locations for long-tailed object detection. In: Proceedings of the Advances in Neural Information Processing Systems, 2024"},{"key":"4416_CR41","doi-asserted-by":"publisher","first-page":"1199","DOI":"10.3390\/electronics12051199","volume":"12","author":"Y Yu","year":"2023","unstructured":"Yu Y, Wang C, Fu Q, et al. Techniques and challenges of image segmentation: a review. Electronics, 2023, 12: 1199","journal-title":"Electronics"},{"key":"4416_CR42","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"X Wang","year":"2024","unstructured":"Wang X, Li S, Kallidromitis K, et al. Hierarchical open-vocabulary universal image segmentation. In: Proceedings of the Advances in Neural Information Processing Systems, 2024"},{"key":"4416_CR43","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"H Wang","year":"2024","unstructured":"Wang H, Li X. Towards generic semi-supervised framework for volumetric medical image segmentation. In: Proceedings of the Advances in Neural Information Processing Systems, 2024"},{"key":"4416_CR44","doi-asserted-by":"publisher","first-page":"103959","DOI":"10.1016\/j.cviu.2024.103959","volume":"241","author":"Q Monnier","year":"2024","unstructured":"Monnier Q, Pouli T, Kpalma K. Survey on fast dense video segmentation techniques. Comput Vision Image Understand, 2024, 241: 103959","journal-title":"Comput Vision Image Understand"},{"key":"4416_CR45","first-page":"2010","volume-title":"Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision","author":"F Wu","year":"2024","unstructured":"Wu F, Marquez-Neila P, Zheng M, et al. Correlation-aware active learning for surgery video segmentation. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, 2024. 2010\u20132020"},{"key":"4416_CR46","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"L Kong","year":"2024","unstructured":"Kong L, Xie S, Hu H, et al. Robodepth: robust out-of-distribution depth estimation under corruptions. In: Proceedings of the Advances in Neural Information Processing Systems, 2024"},{"key":"4416_CR47","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"S Saxena","year":"2024","unstructured":"Saxena S, Herrmann C, Hur J, et al. The surprising effectiveness of diffusion models for optical flow and monocular depth estimation. In: Proceedings of the Advances in Neural Information Processing Systems, 2024"},{"key":"4416_CR48","first-page":"11109","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"R Li","year":"2021","unstructured":"Li R, Zhang S, Wan B, et al. Bipartite graph network with adaptive message passing for unbiased scene graph generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021. 11109\u201311119"},{"key":"4416_CR49","doi-asserted-by":"publisher","first-page":"127052","DOI":"10.1016\/j.neucom.2023.127052","volume":"566","author":"H Li","year":"2024","unstructured":"Li H, Zhu G, Zhang L, et al. Scene graph generation: a comprehensive survey. Neurocomputing, 2024, 566: 127052","journal-title":"Neurocomputing"},{"key":"4416_CR50","doi-asserted-by":"publisher","first-page":"3210","DOI":"10.1109\/TIFS.2024.3360880","volume":"19","author":"M Zhao","year":"2024","unstructured":"Zhao M, Zhang L, Wang W, et al. Adversarial attacks on scene graph generation. IEEE Trans Inform Forensic Secur, 2024, 19: 3210\u20133225","journal-title":"IEEE Trans Inform Forensic Secur"},{"key":"4416_CR51","doi-asserted-by":"publisher","first-page":"103901","DOI":"10.1016\/j.cviu.2023.103901","volume":"239","author":"J Lu","year":"2024","unstructured":"Lu J, Chen L, Guan H, et al. Improving rare relation inferring for scene graph generation using bipartite graph network. Comput Vision Image Understand, 2024, 239: 103901","journal-title":"Comput Vision Image Understand"},{"key":"4416_CR52","first-page":"4629","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"L Zhang","year":"2023","unstructured":"Zhang L, Zhai X, Zhao Z, et al. What if the TV was off? Examining counterfactual reasoning abilities of multi-modal language models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023. 4629\u20134633"},{"key":"4416_CR53","first-page":"2042","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"L Li","year":"2021","unstructured":"Li L, Lei J, Gan Z, et al. Adversarial VQA: a new benchmark for evaluating the robustness of VQA models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021. 2042\u20132051"},{"key":"4416_CR54","first-page":"3081","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","author":"Z Yang","year":"2022","unstructured":"Yang Z, Gan Z, Wang J, et al. An empirical study of GPT-3 for few-shot knowledge-based VQA. In: Proceedings of the AAAI Conference on Artificial Intelligence, 2022. 3081\u20133089"},{"key":"4416_CR55","first-page":"7487","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","author":"P Zhao","year":"2024","unstructured":"Zhao P, Zheng S, Zhao W, et al. Rethinking two-stage referring expression comprehension: a novel grounding and segmentation method modulated by point. In: Proceedings of the AAAI Conference on Artificial Intelligence, 2024. 7487\u20137495"},{"key":"4416_CR56","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"T Nguyen","year":"2024","unstructured":"Nguyen T, Gadre S Y, Ilharco G, et al. Improving multimodal datasets with image captioning. In: Proceedings of the Advances in Neural Information Processing Systems, 2024"},{"key":"4416_CR57","first-page":"4171","volume-title":"Proceedings of the Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","author":"J Devlin","year":"2019","unstructured":"Devlin J, Chang M W, Lee K, et al. BERT: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, 2019. 4171\u20134186"},{"key":"4416_CR58","doi-asserted-by":"publisher","first-page":"147","DOI":"10.3115\/1596374.1596399","volume-title":"Proceedings of the 13th Conference on Computational Natural Language Learning (CoNLL-2009)","author":"L Ratinov","year":"2009","unstructured":"Ratinov L, Roth D. Design challenges and misconceptions in named entity recognition. In: Proceedings of the 13th Conference on Computational Natural Language Learning (CoNLL-2009), 2009. 147\u2013155"},{"key":"4416_CR59","first-page":"280","volume-title":"Proceedings of European Conference on Computer Vision","author":"Y Li","year":"2022","unstructured":"Li Y, Mao H, Girshick R, et al. Exploring plain vision transformer backbones for object detection. In: Proceedings of European Conference on Computer Vision, 2022. 280\u2013296"},{"key":"4416_CR60","first-page":"6748","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"Z Zong","year":"2023","unstructured":"Zong Z, Song G, Liu Y. Detrs with collaborative hybrid assignments training. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023. 6748\u20136758"},{"key":"4416_CR61","doi-asserted-by":"publisher","first-page":"119","DOI":"10.1016\/B978-0-12-821049-9.00003-4","volume-title":"Proceedings of Microscope Image Processing","author":"Q Wu","year":"2023","unstructured":"Wu Q, Castleman K R. Image segmentation. In: Proceedings of Microscope Image Processing, 2023. 119\u2013152"},{"key":"4416_CR62","first-page":"2955","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"J Xu","year":"2023","unstructured":"Xu J, Liu S, Vahdat A, et al. Open-vocabulary panoptic segmentation with text-to-image diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023. 2955\u20132966"},{"key":"4416_CR63","first-page":"1744","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","author":"Z Fang","year":"2024","unstructured":"Fang Z, Guo X, Lin J, et al. An embedding-unleashing video polyp segmentation framework via region linking and scale alignment. In: Proceedings of the AAAI Conference on Artificial Intelligence, 2024. 1744\u20131752"},{"key":"4416_CR64","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"Y Weng","year":"2024","unstructured":"Weng Y, Han M, He H, et al. Mask propagation for efficient video semantic segmentation. In: Proceedings of the Advances in Neural Information Processing Systems, 2024"},{"key":"4416_CR65","doi-asserted-by":"publisher","first-page":"1199","DOI":"10.1109\/TCSVT.2023.3292884","volume":"34","author":"Z Cui","year":"2024","unstructured":"Cui Z, Sheng H, Yang D, et al. Light field depth estimation for non-Lambertian objects via adaptive cross operator. IEEE Trans Circuits Syst Video Technol, 2024, 34: 1199\u20131211","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"key":"4416_CR66","doi-asserted-by":"publisher","first-page":"71","DOI":"10.1145\/3386569.3392377","volume":"39","author":"X Luo","year":"2020","unstructured":"Luo X, Huang J B, Szeliski R, et al. Consistent video depth estimation. ACM Trans Graph, 2020, 39: 71","journal-title":"ACM Trans Graph"},{"key":"4416_CR67","doi-asserted-by":"publisher","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","volume":"39","author":"S Ren","year":"2017","unstructured":"Ren S, He K, Girshick R, et al. Faster R-CNN: towards real-time object detection with region proposal networks. IEEE Trans Pattern Anal Mach Intell, 2017, 39: 1137\u20131149","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"4416_CR68","first-page":"178","volume-title":"Proceedings of European Conference on Computer Vision","author":"J Yang","year":"2022","unstructured":"Yang J, Ang Y Z, Guo Z, et al. Panoptic scene graph generation. In: Proceedings of European Conference on Computer Vision, 2022. 178\u2013196"},{"key":"4416_CR69","first-page":"14974","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Z Shao","year":"2023","unstructured":"Shao Z, Yu Z, Wang M, et al. Prompting large language models with answer heuristics for knowledge-based visual question answering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023. 14974\u201314983"},{"key":"4416_CR70","first-page":"1","volume":"55","author":"P Liu","year":"2023","unstructured":"Liu P, Yuan W, Fu J, et al. Pre-train, prompt, and predict: a systematic survey of prompting methods in natural language processing. ACM Comput Surv, 2023, 55: 1\u201335","journal-title":"ACM Comput Surv"},{"key":"4416_CR71","unstructured":"Fu C, Chen P, Shen Y, et al. MME: a comprehensive evaluation benchmark for multimodal large language models. 2023. ArXiv:2306.13394"},{"key":"4416_CR72","unstructured":"Yu W, Yang Z, Li L, et al. MM-Vet: evaluating large multimodal models for integrated capabilities. 2023. ArXiv:2308.02490"},{"key":"4416_CR73","first-page":"69","volume-title":"Proceedings of European Conference on Computer Vision","author":"L Yu","year":"2016","unstructured":"Yu L, Poirson P, Yang S, et al. Modeling context in referring expressions. In: Proceedings of European Conference on Computer Vision, Amsterdam, 2016. 69\u201385"},{"key":"4416_CR74","first-page":"11","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"J Mao","year":"2016","unstructured":"Mao J, Huang J, Toshev A, et al. Generation and comprehension of unambiguous object descriptions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016. 11\u201320"},{"key":"4416_CR75","first-page":"2641","volume-title":"Proceedings of the IEEE International Conference on Computer Vision","author":"B A Plummer","year":"2015","unstructured":"Plummer B A, Wang L, Cervantes C M, et al. Flickr30k entities: collecting region-to-phrase correspondences for richer image-to-sentence models. In: Proceedings of the IEEE International Conference on Computer Vision, 2015. 2641\u20132649"},{"key":"4416_CR76","first-page":"4566","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"R Vedantam","year":"2015","unstructured":"Vedantam R, Lawrence Zitnick C, Parikh D. CIDEr: consensus-based image description evaluation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2015. 4566\u20134575"},{"key":"4416_CR77","unstructured":"Gao P, Han J, Zhang R, et al. LlaMA-Adapter V2: parameter-efficient visual instruction model. 2023. ArXiv:2304.15010"},{"key":"4416_CR78","unstructured":"Dai W, Li J, Li D, et al. Instructblip: towards general-purpose vision-language models with instruction tuning. 2023. ArXiv:2305.06500"},{"key":"4416_CR79","unstructured":"Li B, Zhang Y, Chen L, et al. Otter: a multi-modal model with in-context instruction tuning. 2023. ArXiv:2305.03726"},{"key":"4416_CR80","unstructured":"Wu W, Yao H, Zhang M, et al. GPT4Vis: what can GPT-4 do for zero-shot visual recognition? 2023. ArXiv:2311.15732, 2023"},{"key":"4416_CR81","volume-title":"Proceedings of the 12th International Conference on Learning Representations","author":"H You","year":"2023","unstructured":"You H, Zhang H, Gan Z, et al. Ferret: refer and ground anything anywhere at any granularity. In: Proceedings of the 12th International Conference on Learning Representations, 2023"},{"key":"4416_CR82","unstructured":"Hurst A, Lerer A, Goucher A P, et al. GPT-4o system card. 2024. ArXiv:2410.21276"},{"key":"4416_CR83","unstructured":"Cao Y, Li S, Liu Y, et al. A comprehensive survey of AI-generated content (AIGC): a history of generative AI from GAN to ChatGPT. 2023. ArXiv:2303.04226"},{"key":"4416_CR84","doi-asserted-by":"crossref","unstructured":"Li C, Wong C, Zhang S, et al. LLaVA-Med: training a large language-and-vision assistant for biomedicine in one day. 2023. ArXiv:2306.00890","DOI":"10.32388\/VLXB6M"},{"key":"4416_CR85","unstructured":"Lu Y, Li C, Liu H, et al. An empirical study of scaling instruct-tuned large multimodal models. 2023. ArXiv:2309.09958"},{"key":"4416_CR86","first-page":"6619","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"K Tang","year":"2019","unstructured":"Tang K, Zhang H, Wu B, et al. Learning to compose dynamic tree structures for visual contexts. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2019. 6619\u20136628"},{"key":"4416_CR87","first-page":"5831","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"R Zellers","year":"2018","unstructured":"Zellers R, Yatskar M, Thomson S, et al. Neural motifs: scene graph parsing with global context. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018. 5831\u20135840"},{"key":"4416_CR88","first-page":"3746","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"X Lin","year":"2020","unstructured":"Lin X, Ding C, Zeng J, et al. GPS-Net: graph property sensing network for scene graph generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2020. 3746\u20133753"},{"key":"4416_CR89","first-page":"700","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops","author":"C Chen","year":"2020","unstructured":"Chen C, Liu M, Meng X, et al. Refinedetlite: a lightweight one-stage object detection framework for CPU-only devices. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops, 2020. 700\u2013701"},{"key":"4416_CR90","doi-asserted-by":"publisher","first-page":"102445","DOI":"10.1016\/j.ecoinf.2023.102445","volume":"79","author":"M Wei","year":"2024","unstructured":"Wei M, Zhan W. YOLO_MRC: a fast and lightweight model for real-time detection and individual counting of Tephritidae pests. Ecol Inf, 2024, 79: 102445","journal-title":"Ecol Inf"},{"key":"4416_CR91","first-page":"2874","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","author":"M Q Le","year":"2024","unstructured":"Le M Q, Nguyen T V, Le T N, et al. MaskDiff: modeling mask distribution with diffusion probabilistic model for few-shot instance segmentation. In: Proceedings of the AAAI Conference on Artificial Intelligence, 2024. 2874\u20132881"},{"key":"4416_CR92","volume-title":"Proceedings of the Advances in Neural Information Processing Systems","author":"S Sun","year":"2024","unstructured":"Sun S, Wang W, Howard A, et al. ReMaX: relaxing for better training on efficient panoptic segmentation. In: Proceedings of the Advances in Neural Information Processing Systems, 2024"},{"key":"4416_CR93","first-page":"13550","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"T D Ngo","year":"2023","unstructured":"Ngo T D, Hua B S, Nguyen K. ISBNet: a 3D point cloud instance segmentation network with instance-aware sampling and box-aware dynamic convolution. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023. 13550\u201313559"},{"key":"4416_CR94","first-page":"23663","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"J He","year":"2023","unstructured":"He J, Li P, Geng Y, et al. Fastinst: a simple query-based model for real-time instance segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023. 23663\u201323672"},{"key":"4416_CR95","first-page":"17819","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"J Hu","year":"2023","unstructured":"Hu J, Huang L, Ren T, et al. You only segment once: towards real-time panoptic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023. 17819\u201317829"}],"container-title":["Science China Information Sciences"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-024-4416-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11432-024-4416-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-024-4416-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,14]],"date-time":"2025-10-14T03:50:52Z","timestamp":1760413852000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11432-024-4416-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,9]]},"references-count":95,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026,2]]}},"alternative-id":["4416"],"URL":"https:\/\/doi.org\/10.1007\/s11432-024-4416-8","relation":{},"ISSN":["1674-733X","1869-1919"],"issn-type":[{"type":"print","value":"1674-733X"},{"type":"electronic","value":"1869-1919"}],"subject":[],"published":{"date-parts":[[2025,10,9]]},"assertion":[{"value":"30 July 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 February 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 April 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 October 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"122103"}}