{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,8]],"date-time":"2026-04-08T05:42:03Z","timestamp":1775626923035,"version":"3.50.1"},"reference-count":45,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2023,4,10]],"date-time":"2023-04-10T00:00:00Z","timestamp":1681084800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,4,10]],"date-time":"2023-04-10T00:00:00Z","timestamp":1681084800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62176027"],"award-info":[{"award-number":["62176027"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2023,8]]},"DOI":"10.1007\/s00530-023-01079-w","type":"journal-article","created":{"date-parts":[[2023,4,10]],"date-time":"2023-04-10T19:02:21Z","timestamp":1681153341000},"page":"1981-1994","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Multi-level network based on transformer encoder for fine-grained image\u2013text matching"],"prefix":"10.1007","volume":"29","author":[{"given":"Lei","family":"Yang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yong","family":"Feng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8302-4435","authenticated-orcid":false,"given":"Mingliang","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiancai","family":"Xiong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yongheng","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Baohua","family":"Qiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,4,10]]},"reference":[{"key":"1079_CR1","doi-asserted-by":"crossref","unstructured":"Rasiwasia N, Costa\u00a0Pereira J, Coviello E, Doyle G, Lanckriet GR, Levy R, Vasconcelos N: A new approach to cross-modal multimedia retrieval. In: Proceedings of the 18th ACM International Conference on Multimedia, pp. 251\u2013260 (2010)","DOI":"10.1145\/1873951.1873987"},{"key":"1079_CR2","doi-asserted-by":"crossref","unstructured":"Feng F, Wang X, Li R: Cross-modal retrieval with correspondence autoencoder. In: Proceedings of the 22nd ACM International Conference on Multimedia, pp. 7\u201316 (2014)","DOI":"10.1145\/2647868.2654902"},{"key":"1079_CR3","doi-asserted-by":"crossref","unstructured":"Klein B, Lev G, Sadeh G, Wolf L: Associating neural word embeddings with deep image representations using fisher vectors. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4437\u20134446 (2015)","DOI":"10.1109\/CVPR.2015.7299073"},{"key":"1079_CR4","doi-asserted-by":"crossref","unstructured":"Karpathy A, Fei-Fei L: Deep visual-semantic alignments for generating image descriptions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3128\u20133137 (2015)","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"1079_CR5","doi-asserted-by":"crossref","unstructured":"Yan F, Mikolajczyk K: Deep correlation for matching images and text. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3441\u20133450 (2015)","DOI":"10.1109\/CVPR.2015.7298966"},{"key":"1079_CR6","unstructured":"Faghri F, Fleet DJ, Kiros JR, Fidler S: Vse++: Improving visual-semantic embeddings with hard negatives. arXiv preprint arXiv:1707.05612 (2017)"},{"key":"1079_CR7","doi-asserted-by":"crossref","unstructured":"Nam H, Ha J-W, Kim J: Dual attention networks for multimodal reasoning and matching. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 299\u2013307 (2017)","DOI":"10.1109\/CVPR.2017.232"},{"key":"1079_CR8","doi-asserted-by":"crossref","unstructured":"Wu Y, Wang S, Song G, Huang Q: Learning fragment self-attention embeddings for image-text matching. In: Proceedings of the 27th ACM International Conference on Multimedia, pp. 2088\u20132096 (2019)","DOI":"10.1145\/3343031.3350940"},{"issue":"1","key":"1079_CR9","doi-asserted-by":"publisher","first-page":"111","DOI":"10.1007\/s00530-016-0532-7","volume":"24","author":"A Jiang","year":"2018","unstructured":"Jiang, A., Li, H., Li, Y., Wang, M.: Learning discriminative representations for semantical crossmodal retrieval. Multimed. Syst. 24(1), 111\u2013121 (2018)","journal-title":"Multimed. Syst."},{"key":"1079_CR10","doi-asserted-by":"crossref","unstructured":"Ge X, Chen F, Jose JM, Ji Z, Wu Z, Liu X: Structured multi-modal feature embedding and alignment for image-sentence retrieval. In: Proceedings of the 29th ACM International Conference on Multimedia, pp. 5185\u20135193 (2021)","DOI":"10.1145\/3474085.3475634"},{"issue":"4","key":"1079_CR11","doi-asserted-by":"publisher","first-page":"2008","DOI":"10.1109\/TIP.2018.2882225","volume":"28","author":"F Huang","year":"2018","unstructured":"Huang, F., Zhang, X.: Zhao: Bi-directional spatial-semantic attention networks for image-text matching. IEEE Trans. Image Proc. 28(4), 2008\u20132020 (2018)","journal-title":"IEEE Trans. Image Proc."},{"key":"1079_CR12","doi-asserted-by":"crossref","unstructured":"Plummer BA, Kordas P, Kiapour MH, Zheng S, Piramuthu R, Lazebnik S: Conditional image-text embedding networks. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 249\u2013264 (2018)","DOI":"10.1007\/978-3-030-01258-8_16"},{"key":"1079_CR13","first-page":"1889","volume":"27","author":"A Karpathy","year":"2014","unstructured":"Karpathy A, Joulin A, Fei-Fei LF: Deep fragment embeddings for bidirectional image sentence mapping. Adv Neural Inf Proc Syst. 27, 1889\u20131897 (2014)","journal-title":"Adv Neural Inf Proc Syst."},{"key":"1079_CR14","doi-asserted-by":"crossref","unstructured":"Lee K-H, Chen X, Hua G, Hu H, He X: Stacked cross attention for image-text matching. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 201\u2013216 (2018)","DOI":"10.1007\/978-3-030-01225-0_13"},{"key":"1079_CR15","doi-asserted-by":"publisher","first-page":"617","DOI":"10.1109\/TIP.2020.3038354","volume":"30","author":"Y Zhang","year":"2020","unstructured":"Zhang, Y., Zhou, W., Wang, M., Tian, Q., Li, H.: Deep relation embedding for cross-modal retrieval. IEEE Trans. Image Proc. 30, 617\u2013627 (2020)","journal-title":"IEEE Trans. Image Proc."},{"key":"1079_CR16","doi-asserted-by":"crossref","unstructured":"Messina N, Amato G, Esuli A, Falchi F, Gennaro C, Marchand-Maillet S: Fine-grained visual textual alignment for cross-modal retrieval using transformer encoders. ACM Transactions on Multimedia Computing, Communications, and Applications (TOMM). 17(4):1\u201323 (2021)","DOI":"10.1145\/3451390"},{"key":"1079_CR17","doi-asserted-by":"crossref","unstructured":"Liu C, Mao Z, Zhang T, Xie H, Wang B: Graph structured network for image-text matching. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10921\u201310930 (2020)","DOI":"10.1109\/CVPR42600.2020.01093"},{"key":"1079_CR18","doi-asserted-by":"crossref","unstructured":"Qu L, Liu M, Cao D, Nie L, Tian Q: Context-aware multi-view summarization network for image-text matching. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 1047\u20131055 (2020)","DOI":"10.1145\/3394171.3413961"},{"key":"1079_CR19","doi-asserted-by":"crossref","unstructured":"Chen H, Ding G, Liu X, Lin Z, Liu J, Han J: Imram: Iterative matching with recurrent attention memory for cross-modal image-text retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12655\u201312663 (2020)","DOI":"10.1109\/CVPR42600.2020.01267"},{"key":"1079_CR20","doi-asserted-by":"publisher","first-page":"2728","DOI":"10.1109\/TIP.2019.2952085","volume":"29","author":"Y Peng","year":"2019","unstructured":"Peng, Y., Qi, J., Zhuo, Y.: Mava: Multi-level adaptive visual-textual alignment by cross-media bi-attention mechanism. IEEE Trans. Image Proc. 29, 2728\u20132741 (2019)","journal-title":"IEEE Trans. Image Proc."},{"key":"1079_CR21","doi-asserted-by":"crossref","unstructured":"Wei X, Zhang T, Li Y, Wu F: Multi-modality cross attention network for image and sentence matching. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10941\u201310950 (2020)","DOI":"10.1109\/CVPR42600.2020.01095"},{"key":"1079_CR22","doi-asserted-by":"crossref","unstructured":"Qu L, Liu M Wu, J, Gao Z, Nie L: Dynamic modality interaction modeling for image-text retrieval. In: Proceedings of the 44th International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 1104\u20131113 (2021)","DOI":"10.1145\/3404835.3462829"},{"issue":"4","key":"1079_CR23","doi-asserted-by":"publisher","first-page":"389","DOI":"10.1007\/s00530-014-0360-6","volume":"20","author":"T Yilmaz","year":"2014","unstructured":"Yilmaz, T., Yazici, A., Kitsuregawa, M.: Relief-mm: effective modality weighting for multimedia information retrieval. Multimed. Syst. 20(4), 389\u2013413 (2014)","journal-title":"Multimed. Syst."},{"issue":"6","key":"1079_CR24","doi-asserted-by":"publisher","first-page":"645","DOI":"10.1007\/s00530-012-0299-4","volume":"20","author":"S Jiang","year":"2014","unstructured":"Jiang, S., Song, X., Huang, Q.: Relative image similarity learning with contextual information for internet cross-media retrieval. Multimed. Syst. 20(6), 645\u2013657 (2014)","journal-title":"Multimed. Syst."},{"key":"1079_CR25","doi-asserted-by":"crossref","unstructured":"Eisenschtat A, Wolf L: Linking image and text with 2-way nets. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4601\u20134611 (2017)","DOI":"10.1109\/CVPR.2017.201"},{"key":"1079_CR26","doi-asserted-by":"crossref","unstructured":"Gu J, Cai J, Joty SR, Niu L, Wang G: Look, imagine and match: Improving textual-visual cross-modal retrieval with generative models. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7181\u20137189 (2018)","DOI":"10.1109\/CVPR.2018.00750"},{"key":"1079_CR27","doi-asserted-by":"crossref","unstructured":"Liu Y, Guo Y, Bakker EM, Lew, MS: Learning a recurrent residual fusion network for multimodal matching. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 4107\u20134116 (2017)","DOI":"10.1109\/ICCV.2017.442"},{"key":"1079_CR28","doi-asserted-by":"crossref","unstructured":"Mithun NC, Panda R, Papalexakis EE, Roy-Chowdhury AK: Webly supervised joint embedding for cross-modal image-text retrieval. In: Proceedings of the 26th ACM International Conference on Multimedia, pp. 1856\u20131864 (2018)","DOI":"10.1145\/3240508.3240712"},{"issue":"2","key":"1079_CR29","doi-asserted-by":"publisher","first-page":"394","DOI":"10.1109\/TPAMI.2018.2797921","volume":"41","author":"L Wang","year":"2018","unstructured":"Wang, L., Li, Y., Huang, J., Lazebnik, S.: Learning two-branch neural networks for image-text matching tasks. IEEE Trans. Patt. Analy. Mach. Intell. 41(2), 394\u2013407 (2018)","journal-title":"IEEE Trans. Patt. Analy. Mach. Intell."},{"key":"1079_CR30","doi-asserted-by":"crossref","unstructured":"Wu Y, Wang S, Huang Q: Learning semantic structure-preserved embeddings for cross-modal retrieval. In: Proceedings of the 26th ACM International Conference on Multimedia, pp. 825\u2013833 (2018)","DOI":"10.1145\/3240508.3240521"},{"key":"1079_CR31","doi-asserted-by":"crossref","unstructured":"Zhang Y, Lu H: Deep cross-modal projection learning for image-text matching. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 686\u2013701 (2018)","DOI":"10.1007\/978-3-030-01246-5_42"},{"key":"1079_CR32","doi-asserted-by":"crossref","unstructured":"Hotelling H: Relations between two sets of variates. In: Breakthroughs in Statistics, pp. 162\u2013190 (1992)","DOI":"10.1007\/978-1-4612-4380-9_14"},{"issue":"12","key":"1079_CR33","doi-asserted-by":"publisher","first-page":"2639","DOI":"10.1162\/0899766042321814","volume":"16","author":"DR Hardoon","year":"2004","unstructured":"Hardoon, D.R., Szedmak, S., Shawe-Taylor, J.: Canonical correlation analysis: an overview with application to learning methods. Neural Comput. 16(12), 2639\u20132664 (2004)","journal-title":"Neural Comput."},{"issue":"2","key":"1079_CR34","first-page":"449","volume":"47","author":"Y Wei","year":"2016","unstructured":"Wei, Y., Zhao, Y., Lu, C., Wei, S., Liu, L., Zhu, Z., Yan, S.: Cross-modal retrieval with cnn visual features: A new baseline. IEEE Trans. Cybernet. 47(2), 449\u2013460 (2016)","journal-title":"IEEE Trans. Cybernet."},{"key":"1079_CR35","doi-asserted-by":"crossref","unstructured":"Zhang L, Ma B, Li G, Huang Q, Tian Q: Multi-networks joint learning for large-scale cross-modal retrieval. In: Proceedings of the 25th ACM International Conference on Multimedia, pp. 907\u2013915 (2017)","DOI":"10.1145\/3123266.3123317"},{"issue":"1","key":"1079_CR36","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3284750","volume":"15","author":"Y Peng","year":"2019","unstructured":"Peng, Y., Qi, J.: Cm-gans: Cross-modal generative adversarial networks for common representation learning. ACM Trans. Multimed. Comput Commun. Appl. (TOMM). 15(1), 1\u201324 (2019)","journal-title":"ACM Trans. Multimed. Comput Commun. Appl. (TOMM)."},{"key":"1079_CR37","doi-asserted-by":"crossref","unstructured":"Wang B, Yang Y, Xu X, Hanjalic A, Shen HT: Adversarial cross-modal retrieval. In: Proceedings of the 25th ACM International Conference on Multimedia, pp. 154\u2013162 (2017)","DOI":"10.1145\/3123266.3123326"},{"key":"1079_CR38","doi-asserted-by":"crossref","unstructured":"Ji Z, Wang H, Han J, Pang Y: Saliency-guided attention network for image-sentence matching. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5754\u20135763 (2019)","DOI":"10.1109\/ICCV.2019.00585"},{"key":"1079_CR39","doi-asserted-by":"crossref","unstructured":"Ma L, Lu Z, Shang L, Li H: Multimodal convolutional neural networks for matching image and sentence. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2623\u20132631 (2015)","DOI":"10.1109\/ICCV.2015.301"},{"key":"1079_CR40","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"1079_CR41","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser L, Polosukhin I: Attention is all you need. In: Advances in Neural Information Processing Systems, pp. 5998\u20136008 (2017)"},{"key":"1079_CR42","unstructured":"Devlin J, Chang M-W, Lee K, Toutanova K: Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"1079_CR43","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1162\/tacl_a_00166","volume":"2","author":"P Young","year":"2014","unstructured":"Young, P., Lai, A., Hodosh, M.: From image descriptions to visual denotations: new similarity metrics for semantic inference over event descriptions. Trans. Assoc. Comput. Linguist. 2, 67\u201378 (2014)","journal-title":"Trans. Assoc. Comput. Linguist."},{"key":"1079_CR44","doi-asserted-by":"crossref","unstructured":"Lin T-Y, Maire M, Belongie S, Hays J, Perona P, Ramanan D, Dollar P, Zitnick CL: Microsoft coco: Common objects in context. In: European Conference on Computer Vision, pp. 740\u2013755 (2014). Springer","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"1079_CR45","first-page":"91","volume":"28","author":"S Ren","year":"2015","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster r-cnn: towards real-time object detection with region proposal networks. Advanc. Neu. Inform. Proc. Syst. 28, 91\u201399 (2015)","journal-title":"Advanc. Neu. Inform. Proc. Syst."}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-023-01079-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-023-01079-w\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-023-01079-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,7,14]],"date-time":"2023-07-14T10:23:56Z","timestamp":1689330236000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-023-01079-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,4,10]]},"references-count":45,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2023,8]]}},"alternative-id":["1079"],"URL":"https:\/\/doi.org\/10.1007\/s00530-023-01079-w","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,4,10]]},"assertion":[{"value":"12 March 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 March 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 April 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}