{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T16:12:54Z","timestamp":1780675974352,"version":"3.54.1"},"reference-count":99,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2023,8,16]],"date-time":"2023-08-16T00:00:00Z","timestamp":1692144000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,8,16]],"date-time":"2023-08-16T00:00:00Z","timestamp":1692144000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100014219","name":"National Science Fund for Distinguished Young Scholars","doi-asserted-by":"publisher","award":["No.62025603"],"award-info":[{"award-number":["No.62025603"]}],"id":[{"id":"10.13039\/501100014219","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["No. U21B2037, No. 62176222, No. 62176223, No. 62176226, No. 62072386, No. 62072387, No. 62072389, and No. 62002305"],"award-info":[{"award-number":["No. U21B2037, No. 62176222, No. 62176223, No. 62176226, No. 62072386, No. 62072387, No. 62072389, and No. 62002305"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100021171","name":"Guangdong Basic and Applied Basic Research Foundation","doi-asserted-by":"crossref","award":["No.2019B1515120049"],"award-info":[{"award-number":["No.2019B1515120049"]}],"id":[{"id":"10.13039\/501100021171","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100015916","name":"Fujian Province Key Laboratory of Special Aquatic Formula Feed","doi-asserted-by":"publisher","award":["No.2021J01002"],"award-info":[{"award-number":["No.2021J01002"]}],"id":[{"id":"10.13039\/501100015916","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100014103","name":"Key Technology Research and Development Program of Shandong","doi-asserted-by":"publisher","award":["No.2022ZD0118201"],"award-info":[{"award-number":["No.2022ZD0118201"]}],"id":[{"id":"10.13039\/100014103","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["No. 20720220068"],"award-info":[{"award-number":["No. 20720220068"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2024,1]]},"DOI":"10.1007\/s11263-023-01871-1","type":"journal-article","created":{"date-parts":[[2023,8,16]],"date-time":"2023-08-16T17:02:02Z","timestamp":1692205322000},"page":"1-19","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":24,"title":["Towards Language-Guided Visual Recognition via Dynamic Convolutions"],"prefix":"10.1007","volume":"132","author":[{"given":"Gen","family":"Luo","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5110-4526","authenticated-orcid":false,"given":"Yiyi","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaoshuai","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yongjian","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yue","family":"Gao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rongrong","family":"Ji","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,8,16]]},"reference":[{"key":"1871_CR1","doi-asserted-by":"crossref","unstructured":"Anderson, P., He, X., Buehler, C., et\u00a0al. (2018). Bottom-up and top-down attention for image captioning and visual question answering. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 6077\u20136086).","DOI":"10.1109\/CVPR.2018.00636"},{"key":"1871_CR2","doi-asserted-by":"crossref","unstructured":"Andreas, J., Rohrbach, M., Darrell, T., et\u00a0al. (2016). Neural module networks. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 39\u201348).","DOI":"10.1109\/CVPR.2016.12"},{"key":"1871_CR3","doi-asserted-by":"crossref","unstructured":"Antol, S., Agrawal, A., Lu, J., et\u00a0al. (2015). Vqa: Visual question answering. In Proceedings of the IEEE international conference on computer vision (pp. 2425\u20132433).","DOI":"10.1109\/ICCV.2015.279"},{"key":"1871_CR4","unstructured":"Bhojanapalli, S., Yun, C., Rawat, A. S., et\u00a0al. (2020). Low-rank bottleneck in multi-head attention models. In International conference on machine learning, PMLR (pp. 864\u2013873)."},{"issue":"19","key":"1871_CR5","doi-asserted-by":"publisher","first-page":"1697","DOI":"10.1016\/j.cub.2007.08.050","volume":"17","author":"B Bonath","year":"2007","unstructured":"Bonath, B., Noesselt, T., Martinez, A., et al. (2007). Neural basis of the ventriloquist illusion. Current Biology, 17(19), 1697\u20131703.","journal-title":"Current Biology"},{"key":"1871_CR6","doi-asserted-by":"publisher","unstructured":"Chen, D. J., Jia, S., Lo, Y. C., et\u00a0al. (2019). See-through-text grouping for referring image segmentation. In 2019 IEEE\/CVF international conference on computer vision (ICCV) (pp. 7453\u20137462). https:\/\/doi.org\/10.1109\/ICCV.2019.00755.","DOI":"10.1109\/ICCV.2019.00755"},{"key":"1871_CR7","doi-asserted-by":"crossref","unstructured":"Chen, Y. C., Li, L., Yu, L., et\u00a0al. (2020). Uniter: Universal image-text representation learning. In European conference on computer vision (pp. 104\u2013120). Springer.","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"1871_CR8","unstructured":"De\u00a0Vries, H., Strub, F., Mary, J., et\u00a0al. (2017). Modulating early visual processing by language. In Advances in neural information processing systems (pp. 6594\u20136604)."},{"key":"1871_CR9","doi-asserted-by":"crossref","unstructured":"Deng, J., Yang, Z., Chen, T., et\u00a0al. (2021). Transvg: End-to-end visual grounding with transformers. arXiv:2104.08541.","DOI":"10.1109\/ICCV48922.2021.00179"},{"key":"1871_CR10","doi-asserted-by":"crossref","unstructured":"Ding, H., Liu, C., Wang, S., et\u00a0al. (2021). Vision-language transformer and query generation for referring segmentation. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 16321\u201316330).","DOI":"10.1109\/ICCV48922.2021.01601"},{"key":"1871_CR11","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2022.3217852","author":"H Ding","year":"2022","unstructured":"Ding, H., Liu, C., Wang, S., et al. (2022). VLT: Vision-language transformer and query generation for referring segmentation. IEEE Transactions on Pattern Analysis and Machine Intelligence. https:\/\/doi.org\/10.1109\/tpami.2022.3217852","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1871_CR12","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., et\u00a0al. (2020). An image is worth 16x16 words: Transformers for image recognition at scale. arXiv:2010.11929."},{"issue":"4","key":"1871_CR13","doi-asserted-by":"publisher","first-page":"419","DOI":"10.1016\/j.cviu.2009.03.008","volume":"114","author":"HJ Escalante","year":"2010","unstructured":"Escalante, H. J., Hern\u00e1ndez, C. A., Gonzalez, J. A., et al. (2010). The segmented and annotated IAPR TC-12 benchmark. Computer Vision and Image Understanding, 114(4), 419\u2013428.","journal-title":"Computer Vision and Image Understanding"},{"key":"1871_CR14","doi-asserted-by":"publisher","unstructured":"Feng, G., Hu, Z., Zhang, L., et\u00a0al. (2021). Encoder fusion network with co-attention embedding for referring image segmentation. https:\/\/doi.org\/10.48550\/ARXIV.2105.01839; https:\/\/arxiv.org\/abs\/2105.01839.","DOI":"10.48550\/ARXIV.2105.01839"},{"key":"1871_CR15","unstructured":"Gan, Z., Chen, Y. C., Li, L., et\u00a0al. (2020). Large-scale adversarial training for vision-and-language representation learning. In NeurIPS."},{"key":"1871_CR16","doi-asserted-by":"crossref","unstructured":"Gao, P., Li, H., Li, S., et\u00a0al. (2018). Question-guided hybrid convolution for visual question answering. In Proceedings of the European conference on computer vision (ECCV) (pp. 469\u2013485).","DOI":"10.1007\/978-3-030-01246-5_29"},{"key":"1871_CR17","doi-asserted-by":"crossref","unstructured":"Goyal, Y., Khot, T., Summers-Stay, D., et\u00a0al. (2017). Making the v in vqa matter: Elevating the role of image understanding in visual question answering. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 6904\u20136913).","DOI":"10.1109\/CVPR.2017.670"},{"key":"1871_CR18","doi-asserted-by":"crossref","unstructured":"He, K., Chen, X., Xie, S., et\u00a0al. (2022). Masked autoencoders are scalable vision learners. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 16000\u201316009).","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"1871_CR19","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., et\u00a0al. (2017). Mask r-cnn. In Proceedings of the IEEE international conference on computer vision (pp. 2961\u20132969).","DOI":"10.1109\/ICCV.2017.322"},{"key":"1871_CR20","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., et\u00a0al. (2016). Deep residual learning for image recognition. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 770\u2013778).","DOI":"10.1109\/CVPR.2016.90"},{"key":"1871_CR21","doi-asserted-by":"crossref","unstructured":"Hu, R., Andreas, J., Darrell, T., et\u00a0al. (2018). Explainable neural computation via stack neural module networks. In Proceedings of the European conference on computer vision (ECCV) (pp. 53\u201369).","DOI":"10.1007\/978-3-030-01234-2_4"},{"key":"1871_CR22","doi-asserted-by":"crossref","unstructured":"Hu, R., Andreas, J., Rohrbach, M., et\u00a0al. (2017a). Learning to reason: End-to-end module networks for visual question answering. In Proceedings of the IEEE international conference on computer vision (pp. 804\u2013813).","DOI":"10.1109\/ICCV.2017.93"},{"key":"1871_CR23","doi-asserted-by":"publisher","unstructured":"Hu, Z., Feng, G., Sun, J., et\u00a0al. (2020). Bi-directional relationship inferring network for referring image segmentation. In 2020 IEEE\/CVF conference on computer vision and pattern recognition (CVPR) (pp. 4423\u20134432). https:\/\/doi.org\/10.1109\/CVPR42600.2020.00448.","DOI":"10.1109\/CVPR42600.2020.00448"},{"key":"1871_CR24","doi-asserted-by":"crossref","unstructured":"Hu, R., Rohrbach, M., Andreas, J., et\u00a0al. (2017b). Modeling relationships in referential expressions with compositional modular networks. In CVPR.","DOI":"10.1109\/CVPR.2017.470"},{"key":"1871_CR25","doi-asserted-by":"crossref","unstructured":"Hu, R., & Singh, A. (2021) Unit: Multimodal multitask learning with a unified transformer. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 1439\u20131449).","DOI":"10.1109\/ICCV48922.2021.00147"},{"key":"1871_CR26","doi-asserted-by":"crossref","unstructured":"Huang, B., Lian, D., Luo, W., et\u00a0al. (2021). Look before you leap: Learning landmark features for one-stage visual grounding. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 16888\u201316897).","DOI":"10.1109\/CVPR46437.2021.01661"},{"key":"1871_CR27","doi-asserted-by":"publisher","unstructured":"Huang, S., Hui, T., Liu, S., et\u00a0al. (2020). Referring image segmentation via cross-modal progressive comprehension. https:\/\/doi.org\/10.48550\/ARXIV.2010.00514; https:\/\/arxiv.org\/abs\/2010.00514.","DOI":"10.48550\/ARXIV.2010.00514"},{"key":"1871_CR28","unstructured":"Hudson, D. A., & Manning, C. D. (2018). Compositional attention networks for machine reasoning. In International conference on learning representations."},{"key":"1871_CR29","unstructured":"Hudson, D., & Manning, C. D. (2019a). Learning by abstraction: The neural state machine. In Advances in neural information processing systems (pp. 5903\u20135916)."},{"key":"1871_CR30","doi-asserted-by":"crossref","unstructured":"Hudson, D. A., & Manning C. D. (2019b). Gqa: A new dataset for real-world visual reasoning and compositional question answering. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 6700\u20136709).","DOI":"10.1109\/CVPR.2019.00686"},{"key":"1871_CR31","doi-asserted-by":"publisher","unstructured":"Hui, T., Liu, S., Huang, S., et\u00a0al. (2020). Linguistic structure guided context modeling for referring image segmentation. https:\/\/doi.org\/10.48550\/ARXIV.2010.00515; https:\/\/arxiv.org\/abs\/2010.00515.","DOI":"10.48550\/ARXIV.2010.00515"},{"key":"1871_CR32","doi-asserted-by":"crossref","unstructured":"Jin, L., Luo, G., Zhou, Y., et\u00a0al. (2023). Refclip: A universal teacher for weakly supervised referring expression comprehension. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 2681\u20132690).","DOI":"10.1109\/CVPR52729.2023.00263"},{"key":"1871_CR33","doi-asserted-by":"publisher","unstructured":"Jing, Y., Kong, T., Wang, W., et\u00a0al. (2021). Locate then segment: A strong pipeline for referring image segmentation. https:\/\/doi.org\/10.48550\/ARXIV.2103.16284; https:\/\/arxiv.org\/abs\/2103.16284.","DOI":"10.48550\/ARXIV.2103.16284"},{"key":"1871_CR34","doi-asserted-by":"crossref","unstructured":"Johnson, J., Hariharan, B., van\u00a0der Maaten, L., et\u00a0al. (2017a). Clevr: A diagnostic dataset for compositional language and elementary visual reasoning. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 2901\u20132910).","DOI":"10.1109\/CVPR.2017.215"},{"key":"1871_CR35","doi-asserted-by":"crossref","unstructured":"Johnson, J., Hariharan, B., Van Der\u00a0Maaten, L., et\u00a0al. (2017b). Inferring and executing programs for visual reasoning. In Proceedings of the IEEE international conference on computer vision (pp. 2989\u20132998).","DOI":"10.1109\/ICCV.2017.325"},{"key":"1871_CR36","doi-asserted-by":"crossref","unstructured":"Kazemzadeh, S., Ordonez, V., Matten, M., et\u00a0al. (2014). Referitgame: Referring to objects in photographs of natural scenes. In EMNLP.","DOI":"10.3115\/v1\/D14-1086"},{"key":"1871_CR37","unstructured":"Kim, J. H., Jun J., Zhang, B. T. (2018). Bilinear attention networks. arXiv:1805.07932."},{"key":"1871_CR38","doi-asserted-by":"publisher","unstructured":"Kim, N., Kim, D., Lan, C., et\u00a0al. (2022). Restr: Convolution-free referring image segmentation using transformers. https:\/\/doi.org\/10.48550\/ARXIV.2203.16768; https:\/\/arxiv.org\/abs\/2203.16768.","DOI":"10.48550\/ARXIV.2203.16768"},{"key":"1871_CR39","unstructured":"Kim, W., Son, B., & Kim, I. (2021). Vilt: Vision-and-language transformer without convolution or region supervision. arXiv:2102.03334."},{"key":"1871_CR40","unstructured":"Kingma, D. P., & Ba, J. (2014). Adam: A method for stochastic optimization. arXiv:1412.6980."},{"issue":"1","key":"1871_CR41","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","volume":"123","author":"R Krishna","year":"2017","unstructured":"Krishna, R., Zhu, Y., Groth, O., et al. (2017). Visual genome: Connecting language and vision using crowdsourced dense image annotations. International Journal of Computer Vision, 123(1), 32\u201373.","journal-title":"International Journal of Computer Vision"},{"key":"1871_CR42","unstructured":"Krizhevsky, A., Sutskever, I., & Hinton, G. E. (2012). Imagenet classification with deep convolutional neural networks. In Advances in neural information processing systems (pp. 1097\u20131105)."},{"key":"1871_CR43","unstructured":"Li, J., Li, D., Savarese, S., et\u00a0al. (2023). Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. arXiv:2301.12597."},{"key":"1871_CR44","doi-asserted-by":"crossref","unstructured":"Lin, T. Y., Maire, M., Belongie, S., et\u00a0al. (2014). Microsoft coco: Common objects in context. In European conference on computer vision (pp. 740\u2013755). Springer.","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"1871_CR45","doi-asserted-by":"crossref","unstructured":"Liu, C., Ding, H., & Jiang, X. (2023). Gres: Generalized referring expression segmentation. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 23592\u201323601).","DOI":"10.1109\/CVPR52729.2023.02259"},{"key":"1871_CR46","doi-asserted-by":"crossref","unstructured":"Liu, C., Lin, Z., Shen, X., et\u00a0al. (2017). Recurrent multimodal interaction for referring image segmentation. In Proceedings of the IEEE international conference on computer vision (pp. 1271\u20131280).","DOI":"10.1109\/ICCV.2017.143"},{"key":"1871_CR47","doi-asserted-by":"crossref","unstructured":"Liu, D., Zhang, H., Wu, F., et\u00a0al. (2019a). Learning to assemble neural module tree networks for visual grounding. In ICCV.","DOI":"10.1109\/ICCV.2019.00477"},{"key":"1871_CR48","doi-asserted-by":"crossref","unstructured":"Liu, R., Liu, C., Bai, Y., et\u00a0al. (2019b). Clevr-ref+: Diagnosing visual reasoning with referring expressions. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 4185\u20134194).","DOI":"10.1109\/CVPR.2019.00431"},{"key":"1871_CR49","doi-asserted-by":"publisher","unstructured":"Liu, S., Hui, T., Huang, S., et\u00a0al. (2021a). Cross-modal progressive comprehension for referring segmentation. https:\/\/doi.org\/10.48550\/ARXIV.2105.07175; https:\/\/arxiv.org\/abs\/2105.07175.","DOI":"10.48550\/ARXIV.2105.07175"},{"key":"1871_CR50","doi-asserted-by":"crossref","unstructured":"Liu, X., Wang, Z., Shao, J., et\u00a0al. (2019c). Improving referring expression grounding with cross-modal attention-guided erasing. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 1950\u20131959).","DOI":"10.1109\/CVPR.2019.00205"},{"key":"1871_CR51","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., et\u00a0al. (2021b). Swin transformer: Hierarchical vision transformer using shifted windows. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 10012\u201310022).","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"1871_CR52","unstructured":"Lu, J., Clark, C., Zellers, R., et\u00a0al. (2022). Unified-io: A unified model for vision, language, and multi-modal tasks. arXiv:2206.08916."},{"key":"1871_CR53","doi-asserted-by":"publisher","unstructured":"Luo, G., Zhou, Y., Ji, R., et\u00a0al. (2020a). Cascade grouped attention network for referring expression segmentation. In Proceedings of the 28th ACM international conference on multimedia. Association for computing machinery, New York, NY, USA, MM \u201920 (pp. 1274\u20131282). https:\/\/doi.org\/10.1145\/3394171.3414006; https:\/\/doi.org\/10.1145\/3394171.3414006.","DOI":"10.1145\/3394171.3414006 10.1145\/3394171.3414006"},{"key":"1871_CR54","doi-asserted-by":"crossref","unstructured":"Luo, G., Zhou, Y., Sun, X., et\u00a0al. (2020b). Multi-task collaborative network for joint referring expression comprehension and segmentation. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR42600.2020.01005"},{"key":"1871_CR55","doi-asserted-by":"publisher","first-page":"3386","DOI":"10.1109\/TIP.2021.3139234","volume":"31","author":"G Luo","year":"2022","unstructured":"Luo, G., Zhou, Y., Sun, X., et al. (2022). Towards lightweight transformer via group-wise transformation for vision-and-language tasks. IEEE Transactions on Image Processing, 31, 3386\u20133398.","journal-title":"IEEE Transactions on Image Processing"},{"key":"1871_CR56","unstructured":"Mao, J., Gan, C., Kohli, P., et\u00a0al. (2019). The neuro-symbolic concept learner: Interpreting scenes, words, and sentences from natural supervision. arXiv:1904.12584."},{"key":"1871_CR57","doi-asserted-by":"crossref","unstructured":"Mao, J., Huang, J., Toshev, A., et\u00a0al. (2016). Generation and comprehension of unambiguous object descriptions. In CVPR.","DOI":"10.1109\/CVPR.2016.9"},{"key":"1871_CR58","doi-asserted-by":"crossref","unstructured":"Mascharka, D., Tran, P., Soklaski, R., et\u00a0al. (2018). Transparency by design: Closing the gap between performance and interpretability in visual reasoning. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 4942\u20134950).","DOI":"10.1109\/CVPR.2018.00519"},{"issue":"5588","key":"1871_CR59","doi-asserted-by":"publisher","first-page":"746","DOI":"10.1038\/264746a0","volume":"264","author":"H McGurk","year":"1976","unstructured":"McGurk, H., & MacDonald, J. (1976). Hearing lips and seeing voices. Nature, 264(5588), 746\u2013748.","journal-title":"Nature"},{"key":"1871_CR60","doi-asserted-by":"crossref","unstructured":"Morency, L. P., Mihalcea, R., Doshi, P. (2011). Towards multimodal sentiment analysis: Harvesting opinions from the web. In Proceedings of the 13th international conference on multimodal interfaces (pp. 169\u2013176).","DOI":"10.1145\/2070481.2070509"},{"key":"1871_CR61","doi-asserted-by":"crossref","unstructured":"Nagaraja, V. K., Morariu, V. I., & Davis, L. S. (2016). Modeling context between objects for referring expression understanding. In ECCV.","DOI":"10.1007\/978-3-319-46493-0_48"},{"key":"1871_CR62","unstructured":"Nguyen, D. K., Goswami, V., & Chen, X. (2020) Revisiting modulated convolutions for visual counting and beyond. arXiv:2004.11883."},{"key":"1871_CR63","doi-asserted-by":"crossref","unstructured":"Pennington, J., Socher, R., & Manning, C. D. (2014). Glove: Global vectors for word representation. In Proceedings of the 2014 conference on empirical methods in natural language processing (EMNLP) (pp. 1532\u20131543).","DOI":"10.3115\/v1\/D14-1162"},{"key":"1871_CR64","doi-asserted-by":"crossref","unstructured":"Perez, E., Strub, F., de\u00a0Vries, H., et\u00a0al. (2018). Film: Visual reasoning with a general conditioning layer. In AAAI.","DOI":"10.1609\/aaai.v32i1.11671"},{"key":"1871_CR65","doi-asserted-by":"crossref","unstructured":"Plummer, B. A., Wang, L., Cervantes, C. M., et\u00a0al. (2015). Flickr30k entities: Collecting region-to-phrase correspondences for richer image-to-sentence models. In Proceedings of the IEEE international conference on computer vision (pp. 2641\u20132649).","DOI":"10.1109\/ICCV.2015.303"},{"key":"1871_CR66","doi-asserted-by":"crossref","unstructured":"Poria, S., Cambria, E., & Gelbukh, A. (2015). Deep convolutional neural network textual features and multiple kernel learning for utterance-level multimodal sentiment analysis. In Proceedings of the 2015 conference on empirical methods in natural language processing (pp. 2539\u20132544).","DOI":"10.18653\/v1\/D15-1303"},{"key":"1871_CR67","doi-asserted-by":"crossref","unstructured":"Qiu, H., Li, H., Wu, Q., et\u00a0al. (2020). Language-aware fine-grained object representation for referring expression comprehension. In Proceedings of the 28th ACM international conference on multimedia (pp. 4171\u20134180).","DOI":"10.1145\/3394171.3413850"},{"key":"1871_CR68","unstructured":"Redmon, J., & Farhadi, A. (2018). Yolov3: An incremental improvement. arXiv:1804.02767."},{"key":"1871_CR69","unstructured":"Ren, S., He, K., Girshick, R., et\u00a0al. (2015). Faster r-cnn: Towards real-time object detection with region proposal networks. arXiv:1506.01497."},{"key":"1871_CR70","doi-asserted-by":"crossref","unstructured":"Sadhu, A., Chen, K., & Nevatia, R. (2019). Zero-shot grounding of objects from natural language queries. In ICCV.","DOI":"10.1109\/ICCV.2019.00479"},{"key":"1871_CR71","unstructured":"Santoro, A., Raposo, D., Barrett, D. G., et\u00a0al. (2017). A simple neural network module for relational reasoning. In Advances in neural information processing systems (pp. 4967\u20134976)."},{"issue":"1","key":"1871_CR72","doi-asserted-by":"publisher","first-page":"147","DOI":"10.1016\/S0926-6410(02)00069-1","volume":"14","author":"L Shams","year":"2002","unstructured":"Shams, L., Kamitani, Y., & Shimojo, S. (2002). Visual illusion induced by sound. Cognitive Brain Research, 14(1), 147\u2013152.","journal-title":"Cognitive Brain Research"},{"key":"1871_CR73","doi-asserted-by":"crossref","unstructured":"Shrestha, R., Kafle, K., & Kanan, C. (2019). Answer them all! toward universal visual question answering models. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 10472\u201310481).","DOI":"10.1109\/CVPR.2019.01072"},{"issue":"4","key":"1871_CR74","doi-asserted-by":"publisher","first-page":"255","DOI":"10.1038\/nrn2331","volume":"9","author":"BE Stein","year":"2008","unstructured":"Stein, B. E., & Stanford, T. R. (2008). Multisensory integration: Current issues from the perspective of the single neuron. Nature Reviews Neuroscience, 9(4), 255\u2013266.","journal-title":"Nature Reviews Neuroscience"},{"key":"1871_CR75","unstructured":"Suarez, J., Johnson, J., & Li, F. F. (2018). Ddrprog: A clevr differentiable dynamic reasoning programmer. arXiv:1803.11361."},{"key":"1871_CR76","doi-asserted-by":"crossref","unstructured":"Sun, J., Luo, G., Zhou, Y., et\u00a0al. (2023). Refteacher: A strong baseline for semi-supervised referring expression comprehension. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 19144\u201319154).","DOI":"10.1109\/CVPR52729.2023.01835"},{"key":"1871_CR77","doi-asserted-by":"crossref","unstructured":"Sun, M., Xiao, J., & Lim, E. G. (2021). Iterative shrinking for referring expression grounding using deep reinforcement learning. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 14060\u201314069).","DOI":"10.1109\/CVPR46437.2021.01384"},{"key":"1871_CR78","unstructured":"Tan, M., & Le, Q. (2019). Efficientnet: Rethinking model scaling for convolutional neural networks. In International conference on machine learning (pp. 6105\u20136114). PMLR."},{"key":"1871_CR79","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., et\u00a0al. (2017). Attention is all you need. In Advances in neural information processing systems (pp. 5998\u20136008)."},{"issue":"2","key":"1871_CR80","doi-asserted-by":"publisher","first-page":"394","DOI":"10.1109\/TPAMI.2018.2797921","volume":"41","author":"L Wang","year":"2018","unstructured":"Wang, L., Li, Y., Huang, J., et al. (2018). Learning two-branch neural networks for image-text matching tasks. IEEE Transactions on Pattern Analysis and Machine Intelligence, 41(2), 394\u2013407.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1871_CR81","doi-asserted-by":"publisher","unstructured":"Wang, Z., Lu, Y., Li, Q., et\u00a0al. (2021a). Cris: Clip-driven referring image segmentation. https:\/\/doi.org\/10.48550\/ARXIV.2111.15174; https:\/\/arxiv.org\/abs\/2111.15174.","DOI":"10.48550\/ARXIV.2111.15174"},{"key":"1871_CR82","unstructured":"Wang, Z., Yu, J., Yu, A. W., et\u00a0al. (2021b). Simvlm: Simple visual language model pretraining with weak supervision. arXiv:2108.10904."},{"key":"1871_CR83","doi-asserted-by":"crossref","unstructured":"Xu, H., Yan, M., Li, C., et\u00a0al. (2021). E2e-vlp: End-to-end vision-language pre-training enhanced by visual learning. arXiv:2106.01804.","DOI":"10.18653\/v1\/2021.acl-long.42"},{"key":"1871_CR84","doi-asserted-by":"publisher","unstructured":"Yang, S., Xia, M., Li, G., et\u00a0al. (2021a). Bottom-up shift and reasoning for referring image segmentation. In 2021 IEEE\/CVF conference on computer vision and pattern recognition (CVPR) (pp. 11261\u201311270). https:\/\/doi.org\/10.1109\/CVPR46437.2021.01111.","DOI":"10.1109\/CVPR46437.2021.01111"},{"key":"1871_CR85","doi-asserted-by":"crossref","unstructured":"Yang, Z., Chen, T., Wang, L., et\u00a0al. (2020). Improving one-stage visual grounding by recursive sub-query construction. arXiv:2008.01059.","DOI":"10.1007\/978-3-030-58568-6_23"},{"key":"1871_CR86","doi-asserted-by":"crossref","unstructured":"Yang, Z., Gong, B., Wang, L., et\u00a0al. (2019). A fast and accurate one-stage approach to visual grounding. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 4683\u20134693).","DOI":"10.1109\/ICCV.2019.00478"},{"key":"1871_CR87","doi-asserted-by":"crossref","unstructured":"Yang, Z., He, X., Gao, J., et\u00a0al. (2016). Stacked attention networks for image question answering. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 21\u201329).","DOI":"10.1109\/CVPR.2016.10"},{"key":"1871_CR88","doi-asserted-by":"publisher","unstructured":"Yang, Z., Wang, J., Tang, Y., et\u00a0al. (2021b). Lavt: Language-aware vision transformer for referring image segmentation. https:\/\/doi.org\/10.48550\/ARXIV.2112.02244; https:\/\/arxiv.org\/abs\/2112.02244.","DOI":"10.48550\/ARXIV.2112.02244"},{"key":"1871_CR89","doi-asserted-by":"publisher","unstructured":"Ye, L., Rochan, M., Liu, Z., et\u00a0al. (2019). Cross-modal self-attention network for referring image segmentation. https:\/\/doi.org\/10.48550\/ARXIV.1904.04745; https:\/\/arxiv.org\/abs\/1904.04745.","DOI":"10.48550\/ARXIV.1904.04745"},{"key":"1871_CR90","unstructured":"Yi, K., Wu, J., Gan, C., et\u00a0al. (2018). Neural-symbolic vqa: Disentangling reasoning from vision and language understanding. In Advances in neural information processing systems (pp. 1031\u20131042)."},{"key":"1871_CR91","doi-asserted-by":"crossref","unstructured":"Yu, F., Tang, J., Yin, W., et\u00a0al. (2021) Ernie-vil: Knowledge enhanced vision-language representations through scene graphs. In Proceedings of the AAAI conference on artificial intelligence (pp. 3208\u20133216).","DOI":"10.1609\/aaai.v35i4.16431"},{"key":"1871_CR92","doi-asserted-by":"crossref","unstructured":"Yu, L., Lin, Z., Shen, X., et\u00a0al. (2018a). Mattnet: Modular attention network for referring expression comprehension. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 1307\u20131315).","DOI":"10.1109\/CVPR.2018.00142"},{"key":"1871_CR93","doi-asserted-by":"crossref","unstructured":"Yu, Z., Yu, J., Cui, Y., et\u00a0al. (2019). Deep modular co-attention networks for visual question answering. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 6281\u20136290).","DOI":"10.1109\/CVPR.2019.00644"},{"key":"1871_CR94","doi-asserted-by":"crossref","unstructured":"Yu, Z., Yu, J., Xiang, C., et\u00a0al. (2018b). Rethinking diversified and discriminative proposal generation for visual grounding. arXiv:1805.03508.","DOI":"10.24963\/ijcai.2018\/155"},{"key":"1871_CR95","unstructured":"Zadeh, A., Zellers, R., Pincus, E., et\u00a0al. (2016). Mosi: Multimodal corpus of sentiment intensity and subjectivity analysis in online opinion videos. arXiv:1606.06259."},{"key":"1871_CR96","doi-asserted-by":"crossref","unstructured":"Zhang, T., Tseng, H. Y., Jiang, L., et\u00a0al. (2020). Text as neural operator: Image manipulation by text instruction. arXiv:2008.04556.","DOI":"10.1145\/3474085.3475343"},{"key":"1871_CR97","unstructured":"Zhou, Y., Ji, R., Luo, G., et\u00a0al. (2021a). A real-time global inference network for one-stage referring expression comprehension. IEEE TNNLS."},{"issue":"2","key":"1871_CR98","doi-asserted-by":"publisher","first-page":"697","DOI":"10.1109\/TPAMI.2019.2956699","volume":"44","author":"Y Zhou","year":"2019","unstructured":"Zhou, Y., Ji, R., Sun, X., et al. (2019). Plenty is plague: Fine-grained learning for visual question answering. IEEE Transactions on Pattern Analysis and Machine Intelligence, 44(2), 697\u2013709.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1871_CR99","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Ren, T., Zhu, C., et\u00a0al. (2021b). Trar: Routing the attention spans in transformer for visual question answering. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 2074\u20132084).","DOI":"10.1109\/ICCV48922.2021.00208"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-023-01871-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-023-01871-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-023-01871-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,22]],"date-time":"2024-01-22T03:06:55Z","timestamp":1705892815000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-023-01871-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,8,16]]},"references-count":99,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2024,1]]}},"alternative-id":["1871"],"URL":"https:\/\/doi.org\/10.1007\/s11263-023-01871-1","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,8,16]]},"assertion":[{"value":"11 October 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"31 July 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 August 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Yongjian Wu is currently the expert researcher and the director of the Youtu Lab, Tencent Co., Ltd. The remaining authors have no relevant financial or non-financial interests to disclose.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"The authors have no relevant ethics approval to disclose.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics Approval"}},{"value":"All authors agreed to participate in this work and made clear contributions.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to Participate"}},{"value":"All authors agreed with the content and that all gave explicit consent to submit and that they obtained consent from the responsible authorities at the institute\/organization where the work has been carried out.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for Publication"}}]}}