{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T05:15:22Z","timestamp":1783401322049,"version":"3.54.6"},"reference-count":57,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2021,1,27]],"date-time":"2021-01-27T00:00:00Z","timestamp":1611705600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,1,27]],"date-time":"2021-01-27T00:00:00Z","timestamp":1611705600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61871278"],"award-info":[{"award-number":["61871278"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Industrial Cluster Collaborative Innovation Project of Chengdu","award":["2016-XT00-00015-GX"],"award-info":[{"award-number":["2016-XT00-00015-GX"]}]},{"name":"Sichuan Science and Technology Program","award":["2018HH0143"],"award-info":[{"award-number":["2018HH0143"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2022,4]]},"DOI":"10.1007\/s11042-020-10466-8","type":"journal-article","created":{"date-parts":[[2021,1,27]],"date-time":"2021-01-27T09:03:02Z","timestamp":1611738182000},"page":"12005-12027","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":8,"title":["Cross-modal multi-relationship aware reasoning for image-text matching"],"prefix":"10.1007","volume":"81","author":[{"given":"Jin","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaohai","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Linbo","family":"Qing","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Luping","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaodong","family":"Luo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2021,1,27]]},"reference":[{"key":"10466_CR1","doi-asserted-by":"crossref","unstructured":"Anderson P, He X, Buehler C, Teney D, Johnson M, Gould S, Zhang L (2018) Bottom-up and top-down attention for image captioning and visual question answering, pp 6077\u20136086, 06","DOI":"10.1109\/CVPR.2018.00636"},{"key":"10466_CR2","doi-asserted-by":"publisher","first-page":"396","DOI":"10.1016\/j.ins.2014.03.128","volume":"279","author":"OA Arqub","year":"2014","unstructured":"Arqub O A, Abo-Hammour ZS (2014) Numerical solution of systems of second-order boundary value problems using continuous genetic algorithm. Inf Sci 279:396\u2013415","journal-title":"Inf Sci"},{"key":"10466_CR3","first-page":"04","volume":"99","author":"J Chen","year":"2019","unstructured":"Chen J, Zhuge H (2019) Extractive summarization of documents with images based on multi-modal rnn. Future Gener Comput Syst 99:04","journal-title":"Future Gener Comput Syst"},{"key":"10466_CR4","unstructured":"Chung J, G\u00fcl\u00e7ehre \u00c7, Cho K, Bengio Y (2014) Empirical evaluation of gated recurrent neural networks on sequence modeling. arXiv:1412.3555"},{"key":"10466_CR5","doi-asserted-by":"crossref","unstructured":"Cornia M, Baraldi L, Tavakoli H R, Cucchiara R (2020) A unified cycle-consistent neural model for text and image retrieval. Multimed Tools Appl 1\u201325, 07","DOI":"10.1007\/s11042-020-09251-4"},{"key":"10466_CR6","unstructured":"Devlin J, Chang M-W, Lee K, Toutanova K (2019) BERT: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies, vol 1 (Long and Short Papers). Association for Computational Linguistics, Minneapolis, pp 4171\u20134186"},{"key":"10466_CR7","unstructured":"Faghri F, Fleet DJ, Kiros JR, Fidler S (2018) Vse++: improving visual-semantic embeddings with hard negatives. In: BMVC"},{"key":"10466_CR8","unstructured":"Frome A, Corrado G S, Shlens J, Bengio S, Dean J, Ranzato M, Mikolov T (2013) Devise: a deep visual-semantic embedding model. In: NIPS, pp 2121\u20132129"},{"key":"10466_CR9","doi-asserted-by":"crossref","unstructured":"Gu J, Cai J, Joty S, Niu L, Wang G (2018) Look, imagine and match: improving textual-visual cross-modal retrieval with generative models. In: 2018 IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp 7181\u20137189","DOI":"10.1109\/CVPR.2018.00750"},{"key":"10466_CR10","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: 2016 IEEE conference on computer vision and pattern recognition (CVPR), pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"10466_CR11","unstructured":"Hou J, Wu X, Qi Y, Zhao W, Luo J, Jia Y (2019) Relational reasoning using prior knowledge for visual captioning. Computer Vision and Pattern Recognition. arXiv:1906.01290"},{"key":"10466_CR12","doi-asserted-by":"crossref","unstructured":"Huang Y, Wang W, Wang L (2017) Instance-aware image and sentence matching with selective multimodal lstm. In: 2017 IEEE conference on computer vision and pattern recognition (CVPR), pp 7254\u20137262","DOI":"10.1109\/CVPR.2017.767"},{"key":"10466_CR13","doi-asserted-by":"crossref","unstructured":"Huang Y, Wu Q, Song C, Wang L (2018) Learning semantic concepts and order for image and sentence matching, pp 6163\u20136171, 06","DOI":"10.1109\/CVPR.2018.00645"},{"key":"10466_CR14","doi-asserted-by":"publisher","first-page":"2008","DOI":"10.1109\/TIP.2018.2882225","volume":"28","author":"F Huang","year":"2019","unstructured":"Huang F, Zhang X, Zhao Z, Li Z (2019) Bi-directional spatial-semantic attention networks for image-text matching. IEEE Trans Image Process 28:2008\u20132020","journal-title":"IEEE Trans Image Process"},{"key":"10466_CR15","doi-asserted-by":"crossref","unstructured":"Karpathy A, Feifei L (2015) Deep visual-semantic alignments for generating image descriptions. In: 2015 IEEE conference on computer vision and pattern recognition (CVPR), pp 3128\u20133137","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"10466_CR16","unstructured":"Kingma D, Adam J B A (2014) A method for stochastic optimization. In: International conference on learning representations, p 12"},{"key":"10466_CR17","unstructured":"Kipf T, Welling M (2017) Semi-supervised classification with graph convolutional networks. arXiv:1609.02907"},{"key":"10466_CR18","doi-asserted-by":"crossref","unstructured":"Klein B, Lev G, Sadeh G, Wolf L (2015) Associating neural word embeddings with deep image representations using fisher vectors. In: IEEE conference on computer vision and pattern recognition (CVPR), pp 4437\u20134446, 06","DOI":"10.1109\/CVPR.2015.7299073"},{"issue":"1","key":"10466_CR19","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","volume":"123","author":"R Krishna","year":"2017","unstructured":"Krishna R, Zhu Y, Groth O, Johnson J, Hata K, Kravitz J, Chen S, Kalantidis Y, Li L, Shamma D A et al (2017) Visual genome: connecting language and vision using crowdsourced dense image annotations. Int J Comput Vis 123(1):32\u201373","journal-title":"Int J Comput Vis"},{"key":"10466_CR20","doi-asserted-by":"crossref","unstructured":"Lee K, Chen X, Hua G, Hu H, He X (2018) Stacked cross attention for image-text matching. In: ECCV. Springer, Cham, pp 212\u2013228","DOI":"10.1007\/978-3-030-01225-0_13"},{"key":"10466_CR21","unstructured":"Li Y, Tarlow D, Brockschmidt M, Zemel R S (2015) Gated graph sequence neural networks. CoRR, arXiv:1511.05493"},{"key":"10466_CR22","doi-asserted-by":"crossref","unstructured":"Li S, Xiao T, Li H, Yang W, Wang X (2017) Identity-aware textual-visual matching with latent co-attention. In: 2017 IEEE international conference on computer vision (ICCV), pp 1908\u20131917","DOI":"10.1109\/ICCV.2017.209"},{"key":"10466_CR23","doi-asserted-by":"crossref","unstructured":"Li K, Zhang Y, Li K, Li Y, Fu Y (2019) Visual semantic reasoning for image-text matching. In: 2019 IEEE\/CVF International conference on computer vision (ICCV), pp 4653\u20134661","DOI":"10.1109\/ICCV.2019.00475"},{"key":"10466_CR24","doi-asserted-by":"crossref","unstructured":"Li L, Gan Z, Cheng Y, Liu J (2019) Relation-aware graph attention network for visual question answering. In: 2019 IEEE\/CVF international conference on computer vision (ICCV), pp 10313\u201310322","DOI":"10.1109\/ICCV.2019.01041"},{"key":"10466_CR25","doi-asserted-by":"crossref","unstructured":"Lin X, Parikh D (2016) Leveraging visual question answering for image-caption ranking 9906:261\u2013277, 10","DOI":"10.1007\/978-3-319-46475-6_17"},{"key":"10466_CR26","doi-asserted-by":"crossref","unstructured":"Lin T -Y, Maire M, Belongie S, Hays J, Perona P, Ramanan D, Doll\u00e1r P, Zitnick C (2014) Microsoft coco: common objects in context, 8693, 04","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"10466_CR27","doi-asserted-by":"publisher","first-page":"365","DOI":"10.1016\/j.patcog.2019.05.008","volume":"93","author":"Y Liu","year":"2019","unstructured":"Liu Y, Guo Y, Liu L, Bakker E M, Lew M S (2019) Cyclematch: a cycle-consistent embedding network for image-text matching. Pattern Recognit 93:365\u2013379, 05","journal-title":"Pattern Recognit"},{"key":"10466_CR28","doi-asserted-by":"crossref","unstructured":"Liu C, Mao Z, Liu A -A, Zhang T, Wang B, Zhang Y (2019) Focus your attention: a bidirectional focal attention network for image-text matching. In: Proceedings of the 27th ACM international conference on multimedia, MM \u201919. Association for Computing Machinery, New York, pp 3\u201311","DOI":"10.1145\/3343031.3350869"},{"key":"10466_CR29","doi-asserted-by":"crossref","unstructured":"Liu C, Mao Z, Zhang T, Xie H, Wang B, Zhang Y (2020) Graph structured network for image-text matching. arXiv:2004.00277","DOI":"10.1109\/CVPR42600.2020.01093"},{"key":"10466_CR30","unstructured":"Lu J, Batra D, Parikh D, Lee S (2019) Vilbert: pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. In: Advances in neural information processing systems, pp 13\u201323"},{"key":"10466_CR31","doi-asserted-by":"crossref","unstructured":"Ma L, Lu Z, Shang L, Li H (2015) Multimodal convolutional neural networks for matching image and sentence. In: 2015 IEEE international conference on computer vision (ICCV), pp 2623\u20132631","DOI":"10.1109\/ICCV.2015.301"},{"key":"10466_CR32","doi-asserted-by":"publisher","first-page":"36","DOI":"10.1016\/j.neucom.2018.11.089","volume":"345","author":"L Ma","year":"2019","unstructured":"Ma L, Jiang W, Jie Z, Wang X (2019) Bidirectional image-sentence retrieval by local and global deep matching. Neurocomputing 345:36\u201344, 02","journal-title":"Neurocomputing"},{"key":"10466_CR33","unstructured":"Messina N, Falchi F, Esuli A, Amato G (2020) Transformer reasoning network for image-text matching and retrieval. arXiv:2004.09144"},{"key":"10466_CR34","unstructured":"Mikolov T, Chen K, Corrado G, Dean J (2013) Efficient estimation of word representations in vector space. arXiv:1301.3781"},{"key":"10466_CR35","doi-asserted-by":"crossref","unstructured":"Nam H, Ha J, Kim J (2017) Dual attention networks for multimodal reasoning and matching. In: 2017 IEEE conference on computer vision and pattern recognition (CVPR), pp 2156\u20132164","DOI":"10.1109\/CVPR.2017.232"},{"key":"10466_CR36","unstructured":"Norcliffebrown W, Vafeias S, Parisot S (2018) Learning conditioned graph structures for interpretable visual question answering. In: NeurIPS, pp 8334\u20138343"},{"key":"10466_CR37","unstructured":"Paszke A, Gross S, Chintala S, Chanan G, Yang E, Devito Z, Lin Z, Desmaison A, Antiga L, Lerer A (2017) Automatic differentiation in pytorch"},{"key":"10466_CR38","unstructured":"Qi D, Su L, Song J, Cui E, Bharti T, Sacheti A (2020) Imagebert: cross-modal pre-training with large-scale weak-supervised image-text data. Comput Vis Pattern Recognit. arXiv:2001.07966"},{"key":"10466_CR39","doi-asserted-by":"publisher","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","volume":"39","author":"S Ren","year":"2015","unstructured":"Ren S, He K, Girshick R, Sun J (2015) Faster r-cnn: towards real-time object detection with region proposal networks. IEEE Trans Pattern Anal Mach Intell 39:1137\u20131149, 06","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"issue":"1","key":"10466_CR40","doi-asserted-by":"publisher","first-page":"61","DOI":"10.1109\/TNN.2008.2005605","volume":"20","author":"F Scarselli","year":"2009","unstructured":"Scarselli F, Gori M, Tsoi A C, Hagenbuchner M, Monfardini G (2009) The graph neural network model. IEEE Trans Neural Netw 20(1):61\u201380","journal-title":"IEEE Trans Neural Netw"},{"key":"10466_CR41","unstructured":"Trott A, Xiong C, Socher R (2018) Interpretable counting for visual question answering. arXiv:1712.08697"},{"key":"10466_CR42","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser L, Polosukhin I (2017) Attention is all you need. In: NIPS"},{"key":"10466_CR43","unstructured":"Velickovic P, Cucurull G, Casanova A, Romero A, Lio P, Bengio Y (2018) Graph attention networks. arXiv:1710.10903"},{"key":"10466_CR44","doi-asserted-by":"crossref","unstructured":"Venugopalan S, Rohrbach M, Donahue J, Mooney R J, Darrell T, Saenko K (2015) Sequence to sequence\u2014video to text. In: 2015 IEEE international conference on computer vision (ICCV), pp 4534\u20134542","DOI":"10.1109\/ICCV.2015.515"},{"key":"10466_CR45","doi-asserted-by":"crossref","unstructured":"Wang L, Li Y, Lazebnik S (2016) Learning deep structure-preserving image-text embeddings. In: 2016 IEEE conference on computer vision and pattern recognition (CVPR), pp 5005\u20135013, 06","DOI":"10.1109\/CVPR.2016.541"},{"key":"10466_CR46","doi-asserted-by":"crossref","unstructured":"Wang Y, Yang H, Qian X, Ma L, Lu J, Li B, Fan X (2019) Position focused attention network for image-text matching. Computation and Language. arXiv:1907.09748","DOI":"10.24963\/ijcai.2019\/526"},{"issue":"2","key":"10466_CR47","doi-asserted-by":"publisher","first-page":"394","DOI":"10.1109\/TPAMI.2018.2797921","volume":"41","author":"L Wang","year":"2019","unstructured":"Wang L, Li Y, Huang J, Lazebnik S (2019) Learning two-branch neural networks for image-text matching tasks. IEEE Trans Pattern Anal Mach Intell 41(2):394\u2013407","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"10466_CR48","doi-asserted-by":"crossref","unstructured":"Wang T, Xu X, Yang Y, Hanjalic A, Shen H, Song J (2019) Matching images and text with multi-modal tensor fusion and re-ranking. In: Proceedings of the 27th ACM international conference on multimedia, pp 12\u201320, 10","DOI":"10.1145\/3343031.3350875"},{"key":"10466_CR49","doi-asserted-by":"crossref","unstructured":"Wang P, Wu Q, Cao J, Shen C, Gao L, Van Den Hengel A (2019) Neighbourhood watch: referring expression comprehension via language-guided graph attention networks. In: 2019 IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp 1960\u20131968","DOI":"10.1109\/CVPR.2019.00206"},{"issue":"2","key":"10466_CR50","first-page":"449","volume":"47","author":"Y Wei","year":"2017","unstructured":"Wei Y, Zhao Y, Lu C, Wei S, Liu L, Zhu Z, Yan S (2017) Cross-modal retrieval with cnn visual features: a new baseline. IEEE Trans Syst Man Cybern 47(2):449\u2013460","journal-title":"IEEE Trans Syst Man Cybern"},{"key":"10466_CR51","first-page":"1","volume":"02","author":"X Xu","year":"2020","unstructured":"Xu X, Wang T, Yang Y, Zuo L, Shen F, Shen H T (2020) Cross-modal attention with semantic consistence for image-text matching. IEEE Trans Neural Netw Learn Syst 02:1\u201314","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"10466_CR52","unstructured":"Yang Z, Qin Z, Yu J, Hu Y (2018) Scene graph reasoning with prior visual relationship for visual question answering. arXiv: Multimedia"},{"key":"10466_CR53","doi-asserted-by":"crossref","unstructured":"Yao T, Pan Y, Li Y, Mei T (2018) Exploring visual relationship for image captioning. In: European conference on computer vision, pp 711\u2013727","DOI":"10.1007\/978-3-030-01264-9_42"},{"issue":"1","key":"10466_CR54","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1162\/tacl_a_00166","volume":"2","author":"P Young","year":"2014","unstructured":"Young P, Lai A, Hodosh M, Hockenmaier J (2014) From image descriptions to visual denotations: new similarity metrics for semantic inference over event descriptions. Trans Assoc Comput Linguist 2(1):67\u201378","journal-title":"Trans Assoc Comput Linguist"},{"key":"10466_CR55","doi-asserted-by":"crossref","unstructured":"Zhang Y, Lu H (2018) Deep cross-modal projection learning for image-text matching. In: The European conference on computer vision (ECCV), pp 707\u2013723","DOI":"10.1007\/978-3-030-01246-5_42"},{"key":"10466_CR56","unstructured":"Zhang Y, Hare J S, Prugelbennett A (2018) Learning to count objects in natural images for visual question answering. arXiv:1802.05766"},{"key":"10466_CR57","doi-asserted-by":"crossref","unstructured":"Zheng Z, Zheng L, Garrett M, Yang Y, Xu M, Shen Y -D (2020) Dual-path convolutional image-text embeddings with instance loss. ACM Trans Multimed Comput Commun Appl 16(2): 1\u201323","DOI":"10.1145\/3383184"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-020-10466-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-020-10466-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-020-10466-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,4,13]],"date-time":"2022-04-13T19:46:43Z","timestamp":1649879203000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-020-10466-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,1,27]]},"references-count":57,"journal-issue":{"issue":"9","published-print":{"date-parts":[[2022,4]]}},"alternative-id":["10466"],"URL":"https:\/\/doi.org\/10.1007\/s11042-020-10466-8","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"value":"1380-7501","type":"print"},{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,1,27]]},"assertion":[{"value":"27 July 2020","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 December 2020","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 December 2020","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 January 2021","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Compliance with Ethical Standards"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"<!--Emphasis Type='Bold' removed-->Conflict of interest"}}]}}