{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,21]],"date-time":"2026-02-21T18:28:27Z","timestamp":1771698507369,"version":"3.50.1"},"reference-count":48,"publisher":"Springer Science and Business Media LLC","issue":"28","license":[{"start":{"date-parts":[[2024,2,5]],"date-time":"2024-02-05T00:00:00Z","timestamp":1707091200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,2,5]],"date-time":"2024-02-05T00:00:00Z","timestamp":1707091200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100012456","name":"National Social Science Fund of China","doi-asserted-by":"publisher","award":["19BYY076"],"award-info":[{"award-number":["19BYY076"]}],"id":[{"id":"10.13039\/501100012456","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-024-18431-5","type":"journal-article","created":{"date-parts":[[2024,2,5]],"date-time":"2024-02-05T06:02:16Z","timestamp":1707112936000},"page":"72043-72062","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Bridging the gap: dual perception attention and local-global similarity fusion for cross-modal image-text matching"],"prefix":"10.1007","volume":"83","author":[{"given":"Xiangyu","family":"Shui","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7217-3109","authenticated-orcid":false,"given":"Zhenfang","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yun","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hongli","family":"Pei","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kefeng","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huaxiang","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,2,5]]},"reference":[{"key":"18431_CR1","doi-asserted-by":"crossref","unstructured":"Wang B, Yang Y , Xu X , Hanjalic A , Shen HT (2017) Adversarial cross-modal retrieval. In: Proceedings of the 25th ACM international conference on multimedia, pp 154\u2013162","DOI":"10.1145\/3123266.3123326"},{"key":"18431_CR2","unstructured":"Faghri F , Fleet DJ , Kiros JR , Fidler S (2017) Vse++: Improving visual-semantic embeddings with hard negatives. arXiv preprint arXiv:1707.05612"},{"key":"18431_CR3","unstructured":"Dutton B (2020) Adversarial canonical correlation analysis. arXiv preprint arXiv:2005.10349"},{"key":"18431_CR4","unstructured":"Kiros R , Salakhutdinov R , Zemel RS (2014) Unifying visual-semantic embeddings with multimodal neural language models. arXiv preprint arXiv:1411.2539"},{"key":"18431_CR5","unstructured":"Simonyan K , Zisserman A (2014) Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556"},{"issue":"6","key":"18431_CR6","doi-asserted-by":"publisher","first-page":"84","DOI":"10.1145\/3065386","volume":"60","author":"A Krizhevsky","year":"2017","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2017) Imagenet classification with deep convolutional neural networks. Commun ACM 60(6):84\u201390","journal-title":"Commun ACM"},{"issue":"8","key":"18431_CR7","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter S, Schmidhuber J (1997) Long short-term memory. Neural Comput 9(8):1735\u20131780","journal-title":"Neural Comput"},{"key":"18431_CR8","doi-asserted-by":"crossref","unstructured":"Cho K , Van\u00a0Merri\u00ebnboer B , Gulcehre C , Bahdanau D , Bougares F , Schwenk H , Bengio Y (2014) Learning phrase representations using rnn encoder-decoder for statistical machine translation. arXiv preprint arXiv:1406.1078","DOI":"10.3115\/v1\/D14-1179"},{"key":"18431_CR9","doi-asserted-by":"crossref","unstructured":"Ji Z , Wang H , Han J , Pang Y (2019) Saliency-guided attention network for image-sentence matching. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 5754\u20135763","DOI":"10.1109\/ICCV.2019.00585"},{"key":"18431_CR10","doi-asserted-by":"crossref","unstructured":"Lee K-H , Chen X , Hua G , Hu H , He X (2018) Stacked cross attention for image-text matching. In: Proceedings of the European conference on computer vision (ECCV), pp 201\u2013216","DOI":"10.1007\/978-3-030-01225-0_13"},{"key":"18431_CR11","doi-asserted-by":"crossref","unstructured":"Wang Y , Yang H , Qian X , Ma L , Lu J , Li B , Fan X (2019) Position focused attention network for image-text matching. arXiv preprint arXiv:1907.09748","DOI":"10.24963\/ijcai.2019\/526"},{"key":"18431_CR12","doi-asserted-by":"crossref","unstructured":"Liu C , Mao Z , Liu A-A , Zhang T , Wang B , Zhang Y (2019) Focus your attention: A bidirectional focal attention network for image-text matching. In: Proceedings of the 27th ACM international conference on multimedia, pp 3\u201311","DOI":"10.1145\/3343031.3350869"},{"key":"18431_CR13","doi-asserted-by":"crossref","unstructured":"Ge X , Chen F , Xu S , Tao F , Jose JM (2023) Cross-modal semantic enhanced interaction for image-sentence retrieval. In: Proceedings of the IEEE\/CVF winter conference on applications of computer vision, pp 1022\u20131031","DOI":"10.1109\/WACV56688.2023.00108"},{"key":"18431_CR14","doi-asserted-by":"crossref","unstructured":"Liu C , Mao Z , Zhang T , Xie H , Wang B , Zhang Y (2020) Graph structured network for image-text matching. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 10921\u201310930","DOI":"10.1109\/CVPR42600.2020.01093"},{"key":"18431_CR15","doi-asserted-by":"crossref","unstructured":"Li Z , Guo C , Feng Z , Hwang J-N , Xue X (2022) Multi-view visual semantic embedding. In: IJCAI, vol 2, p 7","DOI":"10.24963\/ijcai.2022\/158"},{"key":"18431_CR16","doi-asserted-by":"crossref","unstructured":"Pan Z , Wu F , Zhang B (2023) Fine-grained image-text matching by cross-modal hard aligning network. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 19275\u201319284","DOI":"10.1109\/CVPR52729.2023.01847"},{"key":"18431_CR17","doi-asserted-by":"publisher","first-page":"21847","DOI":"10.1109\/ACCESS.2020.2969808","volume":"8","author":"Z Li","year":"2020","unstructured":"Li Z, Ling F, Zhang C, Ma H (2020) Combining global and local similarity for cross-media retrieval. IEEE Access 8:21847\u201321856","journal-title":"IEEE Access"},{"key":"18431_CR18","doi-asserted-by":"crossref","unstructured":"Diao H, Zhang Y, Ma L, Lu H (2021) Similarity reasoning and filtration for image-text matching. In: Proceedings of the AAAI conference on artificial intelligence vol 35, pp 1218\u20131226","DOI":"10.1609\/aaai.v35i2.16209"},{"issue":"7","key":"18431_CR19","doi-asserted-by":"publisher","first-page":"2866","DOI":"10.1109\/TCSVT.2020.3030656","volume":"31","author":"K Wen","year":"2020","unstructured":"Wen K, Gu X, Cheng Q (2020) Learning dual semantic relations with graph attention for image-text matching. IEEE Trans Circuits Syst Video Technol 31(7):2866\u20132879","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"key":"18431_CR20","doi-asserted-by":"crossref","unstructured":"Zhang Q , Lei Z , Zhang Z , Li SZ (2020) t-aware attention network for image-text retrieval. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 3536\u20133545","DOI":"10.1109\/CVPR42600.2020.00359"},{"key":"18431_CR21","doi-asserted-by":"crossref","unstructured":"Wang H , Zhang Y , Ji Z , Pang Y , Ma L (2020) Consensus-aware visual-semantic embedding for image-text matching. In: Computer Vision\u2013ECCV 2020: 16th European conference, glasgow, UK, August 23\u201328, 2020, Proceedings, Part XXIV 16, pp 18\u201334. Springer","DOI":"10.1007\/978-3-030-58586-0_2"},{"key":"18431_CR22","doi-asserted-by":"crossref","unstructured":"Zhang K , Mao Z , Wang Q , Zhang Y (2022) Negative-aware attention framework for image-text matching. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 15661\u201315670","DOI":"10.1109\/CVPR52688.2022.01521"},{"key":"18431_CR23","doi-asserted-by":"crossref","unstructured":"Wang L , Li Y , Lazebnik S (2016) Learning deep structure-preserving image-text embeddings. In: Proceedings of the IEEE conference on computer vision and pattern recognition (CVPR)","DOI":"10.1109\/CVPR.2016.541"},{"key":"18431_CR24","doi-asserted-by":"crossref","unstructured":"Sarafianos N , Xu X , Kakadiaris IA (2019) Adversarial representation learning for text-to-image matching. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 5814\u20135824","DOI":"10.1109\/ICCV.2019.00591"},{"issue":"2","key":"18431_CR25","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3383184","volume":"16","author":"Z Zheng","year":"2020","unstructured":"Zheng Z, Zheng L, Garrett M, Yang Y, Xu M, Shen Y-D (2020) Dual-path convolutional image-text embeddings with instance loss. ACM Trans Multimed Comput Commun Appl (TOMM) 16(2):1\u201323","journal-title":"ACM Trans Multimed Comput Commun Appl (TOMM)"},{"key":"18431_CR26","unstructured":"Jia C , Yang Y , Xia Y , Chen Y-T , Parekh Z , Pham H , Le Q , Sung Y-H , Li Z , Duerig T (2021) Scaling up visual and vision-language representation learning with noisy text supervision. In: International conference on machine learning, pp 4904\u20134916. PMLR"},{"key":"18431_CR27","doi-asserted-by":"crossref","unstructured":"Huang Y , Wu Q , Song C , Wang L (2018) Learning semantic concepts and order for image and sentence matching. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 6163\u20136171","DOI":"10.1109\/CVPR.2018.00645"},{"key":"18431_CR28","doi-asserted-by":"crossref","unstructured":"Chen H , Ding G , Liu X , Lin Z , Liu J , Han J (2020) Imram: Iterative matching with recurrent attention memory for cross-modal image-text retrieval. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 12655\u2013 12663","DOI":"10.1109\/CVPR42600.2020.01267"},{"key":"18431_CR29","doi-asserted-by":"crossref","unstructured":"Zhuge M , Gao D , Fan D-P , Jin L , Chen B , Zhou H , Qiu M , Shao L (2021) Kaleido-bert: Vision-language pre-training on fashion domain. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 12647\u2013 12657","DOI":"10.1109\/CVPR46437.2021.01246"},{"key":"18431_CR30","doi-asserted-by":"crossref","unstructured":"Ji Z , Chen K , Wang H (2020) Step-wise hierarchical alignment network for image-text matching. arXiv preprint arXiv:2106.06509","DOI":"10.24963\/ijcai.2021\/106"},{"key":"18431_CR31","doi-asserted-by":"crossref","unstructured":"Wei X , Zhang T , Li Y , Zhang Y , Wu F (2020) Multi-modality cross attention network for image and sentence matching. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 10941\u2013 10950","DOI":"10.1109\/CVPR42600.2020.01095"},{"key":"18431_CR32","doi-asserted-by":"crossref","unstructured":"Li L, Gan Z, Cheng Y, Liu J (2019) Relation-aware graph attention network for visual question answering. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 10313\u2013 10322","DOI":"10.1109\/ICCV.2019.01041"},{"key":"18431_CR33","doi-asserted-by":"crossref","unstructured":"Nam H , Ha J-W, Kim J (2017) Dual attention networks for multimodal reasoning and matching. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 299\u2013 307","DOI":"10.1109\/CVPR.2017.232"},{"issue":"2","key":"18431_CR34","doi-asserted-by":"publisher","first-page":"394","DOI":"10.1109\/TPAMI.2018.2797921","volume":"41","author":"L Wang","year":"2018","unstructured":"Wang L, Li Y, Huang J, Lazebnik S (2018) Learning two-branch neural networks for image-text matching tasks. IEEE Trans Pattern Anal Mach Intell 41(2):394\u2013407","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"issue":"2","key":"18431_CR35","doi-asserted-by":"publisher","first-page":"103223","DOI":"10.1016\/j.ipm.2022.103223","volume":"60","author":"Z Zhu","year":"2023","unstructured":"Zhu Z, Zhang D, Li L, Li K, Qi J, Wang W, Zhang G, Liu P (2023) Knowledge-guided multi-granularity gcn for absa. Inform Process Manag 60(2):103223","journal-title":"Inform Process Manag"},{"key":"18431_CR36","unstructured":"Devlin J, Chang M-W, Lee K, Toutanova K (2018) Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805"},{"key":"18431_CR37","unstructured":"Lu J, Batra D, Parikh D, Lee S (2019) Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. Adv Neural Inform Process Syst 32"},{"key":"18431_CR38","doi-asserted-by":"crossref","unstructured":"Chen Y-C, Li L, Yu L, El\u00a0Kholy A, Ahmed F , Gan Z , Cheng Y , Liu J (2020) Uniter: Universal image-text representation learning. In: European conference on computer vision, pp 104\u2013120. Springer","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"18431_CR39","unstructured":"Ren S, He K, Girshick R, Sun J (2015) Faster r-cnn: Towards real-time object detection with region proposal networks. Adv Neural Inform Process Syst 28"},{"key":"18431_CR40","doi-asserted-by":"crossref","unstructured":"Plummer BA , Wang L , Cervantes CM, Caicedo JC, Hockenmaier J, Lazebnik S (2015) Flickr30k entities: Collecting region-to-phrase correspondences for richer image-to-sentence models. In: Proceedings of the IEEE international conference on computer vision, pp 2641\u2013 2649","DOI":"10.1109\/ICCV.2015.303"},{"key":"18431_CR41","doi-asserted-by":"crossref","unstructured":"Lin T-Y, Maire M, Belongie S , Hays J , Perona P , Ramanan D , Doll\u00e1r P , Zitnick CL (2014) Microsoft coco: Common objects in context. In: European conference on computer vision, pp 740\u2013 755. Springer","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"18431_CR42","doi-asserted-by":"crossref","unstructured":"Wang T , Xu X , Yang Y , Hanjalic A , Shen HT , Song J (2019) Matching images and text with multi-modal tensor fusion and re-ranking. In: Proceedings of the 27th ACM international conference on multimedia, pp 12\u2013 20","DOI":"10.1145\/3343031.3350875"},{"key":"18431_CR43","doi-asserted-by":"crossref","unstructured":"Li K , Zhang Y , Li K , Li Y , Fu Y (2022) Image-text embedding learning via visual and textual semantic reasoning. IEEE Trans Pattern Anal Mach Intell","DOI":"10.1109\/TPAMI.2022.3148470"},{"key":"18431_CR44","doi-asserted-by":"crossref","unstructured":"Fu Z , Mao Z , Song Y , Zhang Y (2023) Learning semantic relationship among instances for image-text matching. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp 15159\u2013 15168","DOI":"10.1109\/CVPR52729.2023.01455"},{"issue":"4","key":"18431_CR45","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3499027","volume":"18","author":"Y Cheng","year":"2022","unstructured":"Cheng Y, Zhu X, Qian J, Wen F, Liu P (2022) Cross-modal graph matching network for image-text retrieval. ACM Trans Multimed Comput Commun Appl (TOMM) 18(4):1\u201323","journal-title":"ACM Trans Multimed Comput Commun Appl (TOMM)"},{"key":"18431_CR46","doi-asserted-by":"publisher","first-page":"36","DOI":"10.1016\/j.neucom.2018.11.089","volume":"345","author":"L Ma","year":"2019","unstructured":"Ma L, Jiang W, Jie Z, Wang X (2019) Bidirectional image-sentence retrieval by local and global deep matching. Neurocomputing 345:36\u201344","journal-title":"Neurocomputing"},{"issue":"12","key":"18431_CR47","doi-asserted-by":"publisher","first-page":"5412","DOI":"10.1109\/TNNLS.2020.2967597","volume":"31","author":"X Xu","year":"2020","unstructured":"Xu X, Wang T, Yang Y, Zuo L, Shen F, Shen HT (2020) Cross-modal attention with semantic consistence for image-text matching. IEEE Trans Neural Netw Learn Syst 31(12):5412\u20135425","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"18431_CR48","doi-asserted-by":"crossref","unstructured":"Zhang H, Mao Z, Zhang K, Zhang Y (2022) Show your faith: Cross-modal confidence-aware network for image-text matching. In: Proceedings of the AAAI Conference on Artificial Intelligence vol 36, pp 3262\u20133270","DOI":"10.1609\/aaai.v36i3.20235"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-024-18431-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-024-18431-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-024-18431-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,30]],"date-time":"2024-07-30T17:09:41Z","timestamp":1722359381000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-024-18431-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,2,5]]},"references-count":48,"journal-issue":{"issue":"28","published-online":{"date-parts":[[2024,8]]}},"alternative-id":["18431"],"URL":"https:\/\/doi.org\/10.1007\/s11042-024-18431-5","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,2,5]]},"assertion":[{"value":"14 November 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 January 2024","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 January 2024","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 February 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of Interest"}}]}}