{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T20:01:18Z","timestamp":1781640078175,"version":"3.54.5"},"reference-count":43,"publisher":"Springer Science and Business Media LLC","issue":"23","license":[{"start":{"date-parts":[[2024,9,11]],"date-time":"2024-09-11T00:00:00Z","timestamp":1726012800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,9,11]],"date-time":"2024-09-11T00:00:00Z","timestamp":1726012800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100007129","name":"Natural Science Foundation of Shandong Province","doi-asserted-by":"publisher","award":["ZR2023MF031"],"award-info":[{"award-number":["ZR2023MF031"]}],"id":[{"id":"10.13039\/501100007129","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62102186"],"award-info":[{"award-number":["62102186"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62171209"],"award-info":[{"award-number":["62171209"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1007\/s10489-024-05823-1","type":"journal-article","created":{"date-parts":[[2024,9,11]],"date-time":"2024-09-11T07:02:39Z","timestamp":1726038159000},"page":"12230-12245","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Cross-modality interaction reasoning for enhancing vision-language pre-training in image-text retrieval"],"prefix":"10.1007","volume":"54","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2660-1050","authenticated-orcid":false,"given":"Tao","family":"Yao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shouyong","family":"Peng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lili","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ying","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yujuan","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,9,11]]},"reference":[{"key":"5823_CR1","doi-asserted-by":"crossref","unstructured":"Chen H, Ding G, Liu X, Lin Z, Liu J, Han J (2020) Imram: iterative matching with recurrent attention memory for cross-modal image-text retrieval. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 12655\u201312663","DOI":"10.1109\/CVPR42600.2020.01267"},{"issue":"4","key":"5823_CR2","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3499027","volume":"18","author":"Y Cheng","year":"2022","unstructured":"Cheng Y, Zhu X, Qian J, Wen F, Liu P (2022) Cross-modal graph matching network for image-text retrieval. ACM Trans Multimed Comput Commun Appl 18(4):1\u201323","journal-title":"ACM Trans Multimed Comput Commun Appl"},{"key":"5823_CR3","doi-asserted-by":"crossref","unstructured":"Dey R, Salem F (2017) Gate-variants of gated recurrent unit (gru) neural networks. In: 2017 IEEE 60th international midwest symposium on circuits and systems, pp 1597\u20131600","DOI":"10.1109\/MWSCAS.2017.8053243"},{"key":"5823_CR4","doi-asserted-by":"crossref","unstructured":"Diao H, Zhang Y, Ma L, Lu H (2021) Similarity reasoning and filtration for image-text retrieval. In: Proceedings of the AAAI conference on artificial intelligence, pp 1218-1226","DOI":"10.1609\/aaai.v35i2.16209"},{"key":"5823_CR5","unstructured":"Faghri F, Fleet D, Kiros J, Fidler S (2018) Vse++: improving visual-semantic embeddings with hard negatives. Proceedings of the British machine vision conference"},{"issue":"5","key":"5823_CR6","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3580501","volume":"19","author":"D Feng","year":"2023","unstructured":"Feng D, He X, Peng Y (2023) Mkvse: multimodal knowledge enhanced visual-semantic embedding for image-text retrieval. ACM Trans Multimed Comput Commun Appl 19(5):1\u201321","journal-title":"ACM Trans Multimed Comput Commun Appl"},{"key":"5823_CR7","doi-asserted-by":"crossref","unstructured":"Fu Z, Mao Z, Song Y, Zhang Y (2023) Learning semantic relationship among instances for image-text matching. In: Proceedings of the IEEE\/CVF conference on computer vision pattern recognition, pp 15159\u201315168","DOI":"10.1109\/CVPR52729.2023.01455"},{"key":"5823_CR8","doi-asserted-by":"crossref","unstructured":"Ge X, Chen F, Jose J, Ji Z, Wu Z, Liu X (2021) Structured multi-modal feature embedding and alignment for image-sentence retrieval. In: Proceedings of the 29th ACM international conference on multimedia, pp 5185\u20135193","DOI":"10.1145\/3474085.3475634"},{"key":"5823_CR9","doi-asserted-by":"crossref","unstructured":"Hu Z, Luo Y, Lin J, Yan Y, Chen J (2019) Multi-level visual-semantic alignments with relation-wise dual attention network for image and text matching. In: International joint conferences on artificial intelligence, pp 789\u2013795","DOI":"10.24963\/ijcai.2019\/111"},{"key":"5823_CR10","doi-asserted-by":"crossref","unstructured":"Ji Z, Chen K, Wang H (2021) Step-wise hierarchical alignment network for image-text retrieval. Proceedings of the thirtieth international joint conference on artificial intelligence","DOI":"10.24963\/ijcai.2021\/106"},{"key":"5823_CR11","unstructured":"Karpathy A, Joulin A, Fei-Fei L (2014) Deep fragment embeddings for bidirectional image sentence mapping. Advances in neural information processing systems, pp 27"},{"key":"5823_CR12","unstructured":"Kim W, Son B, Kim I (2021) Vilt: vision-and-language transformer without convolution or region supervision. In: International conference on machine learning, pp 5583\u20135594"},{"key":"5823_CR13","doi-asserted-by":"crossref","unstructured":"Kim D, Kim N, Kwak S (2023) Improving cross-modal retrieval with set of diverse embeddings. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 23422\u201323431","DOI":"10.1109\/CVPR52729.2023.02243"},{"key":"5823_CR14","doi-asserted-by":"crossref","unstructured":"Klein B, Lev G, Sadeh G, Wolf L (2015) Associating neural word embeddings with deep image representations using fisher vectors. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4437\u20134446","DOI":"10.1109\/CVPR.2015.7299073"},{"key":"5823_CR15","doi-asserted-by":"crossref","unstructured":"Lee K, Chen X, Hua G, Hu H, He X (2018) Stacked cross attention for image-text retrieval. In: Proceedings of the European conference on computer vision, pp 201\u2013216","DOI":"10.1007\/978-3-030-01225-0_13"},{"key":"5823_CR16","doi-asserted-by":"crossref","unstructured":"Li J, Niu L, Zhang L (2022) Action-aware embedding enhancement for image-text retrieval. In: Proceedings of the AAAI conference on artificial intelligence, pp 1323\u20131331","DOI":"10.1609\/aaai.v36i2.20020"},{"key":"5823_CR17","unstructured":"Li J, Li D, Xiong C, Hoi S (2022) Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International conference on machine learning, pp 12888\u201312900"},{"key":"5823_CR18","doi-asserted-by":"crossref","unstructured":"Li K, Zhang Y, Li K, Li Y, Fu Y (2019) Visual semantic reasoning for image-text retrieval. In: Proceedings of the IEEE\/CVF International conference on computer vision, pp 4654\u20134662","DOI":"10.1109\/ICCV.2019.00475"},{"key":"5823_CR19","doi-asserted-by":"crossref","unstructured":"Lin T, Maire M, Belongie S, Hays J, Perona P, Ramanan D, Doll\u00e1r P, Zitnick C (2014) Microsoft coco: common objects in context. In: Computer vision\u2013ECCV, pp 740\u2013755","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"5823_CR20","doi-asserted-by":"crossref","unstructured":"Liu C, Mao Z, Liu A, Zhang T, Wang B, Zhang Y (2019) Focus your attention: a bidirectional focal attention network for image-text retrieval. In: Proceedings of the 27th ACM international conference on multimedia, pp 3\u201311","DOI":"10.1145\/3343031.3350869"},{"key":"5823_CR21","doi-asserted-by":"crossref","unstructured":"Liu Y, Liu H, Wang H, Liu M (2022) Regularizing visual semantic embedding with contrastive learning for image-text retrieval. IEEE Signal Processing Letters pp 29:1332\u20131336","DOI":"10.1109\/LSP.2022.3178899"},{"key":"5823_CR22","doi-asserted-by":"crossref","unstructured":"Li K, Zhang Y, Li K, Li Y, Fu Y (2022) Image-text embedding learning via visual and textual semantic reasoning. IEEE transactions on pattern analysis and machine intelligence pp 45:641\u2013656","DOI":"10.1109\/TPAMI.2022.3148470"},{"key":"5823_CR23","doi-asserted-by":"crossref","unstructured":"Long S, Han S, Wan X, Poon J (2022) Gradual: Graph-based dual-modal representation for image-text matching. In: Proceedings of the IEEE\/CVF winter conference on applications of computer vision, pp 3459\u20133468","DOI":"10.1109\/WACV51458.2022.00252"},{"key":"5823_CR24","unstructured":"Lu J, Batra D, Parikh D, Lee S (2019) Vilbert: pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. Advances in neural information processing systems, vol 32"},{"key":"5823_CR25","unstructured":"Malinowski M, Fritz M (2014) A multi-world approach to question answering about real-world scenes based on uncertain input. Advances in neural information processing systems vol 27"},{"key":"5823_CR26","doi-asserted-by":"crossref","unstructured":"Nie L, Qu L, Meng D, Zhang M, Tian Q, Bimbo A (2022) Search-oriented micro-video captioning. In: Proceedings of the 30th ACM international conference on multimedia, pp 3234\u20133243","DOI":"10.1145\/3503161.3548180"},{"key":"5823_CR27","doi-asserted-by":"crossref","unstructured":"Pan Z, Wu F, Zhang B (2023) Fine-grained image-text matching by cross-modal hard aligning network. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 19275\u201319284","DOI":"10.1109\/CVPR52729.2023.01847"},{"key":"5823_CR28","doi-asserted-by":"crossref","unstructured":"Peng L, Qian J, Wang C, Liu B, Dong Y (2023) Swin transformer-based supervised hashing. Applied intelligence, pp 1\u201313","DOI":"10.1007\/s10489-022-04410-6"},{"key":"5823_CR29","doi-asserted-by":"crossref","unstructured":"Plummer B, Wang L, Cervantes C, Caicedo J, Hockenmaier J, Lazebnik S (2015) Flickr30k entities: collecting region-to-phrase correspondences for richer image-to-sentence models. In: Proceedings of the IEEE international conference on computer vision, pp. 2641\u20132649","DOI":"10.1109\/ICCV.2015.303"},{"key":"5823_CR30","unstructured":"Radford A, Kim J, Hallacy C, Ramesh A, Goh G, Agarwal S,Sastry G, Askell A, Mishkin P, Clark J, et\u00a0al (2021) Learning transferable visual models from natural language supervision. In: International conference on machine learning, pp 8748\u20138763"},{"key":"5823_CR31","doi-asserted-by":"crossref","unstructured":"Shu Z, Li L, Yu J, Zhang D, Yu Z, Wu X (2023) Online supervised collective matrix factorization hashing for cross-modal retrieval. Applied intelligence, pp 14201\u201314218","DOI":"10.1007\/s10489-022-04189-6"},{"key":"5823_CR32","unstructured":"Simonyan K, Zisserman A (2014) Very deep convolutional networks for large-scale image recognition"},{"key":"5823_CR33","doi-asserted-by":"crossref","unstructured":"Wang H, Zhang Y, Ji Z, Pang Y, Ma L (2020) Consensus-aware visual-semantic embedding for image-text retrieval. In: Computer vision\u2013ECCV, pp 18\u201334","DOI":"10.1007\/978-3-030-58586-0_2"},{"key":"5823_CR34","doi-asserted-by":"crossref","unstructured":"Wang J, Zhou P, Shou M, Yan S (2023) Position-guided text prompt for vision-language pre-training. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 23242\u201323251","DOI":"10.1109\/CVPR52729.2023.02226"},{"key":"5823_CR35","doi-asserted-by":"crossref","unstructured":"Wang S, Wang R, Yao Z, Shan S, Chen X (2020) Cross-modal scene graph matching for relationship-aware image-text retrieval. In: Proceedings of the IEEE\/CVF winter conference on applications of computer vision, pp 1508\u20131517","DOI":"10.1109\/WACV45572.2020.9093614"},{"key":"5823_CR36","doi-asserted-by":"crossref","unstructured":"Wang Z, Liu X, Li H, Sheng L, Yan J, Wang X, Shao J (2019) Camp: cross-modal adaptive message passing for text-image retrieval. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 5764\u20135773","DOI":"10.1109\/ICCV.2019.00586"},{"key":"5823_CR37","doi-asserted-by":"crossref","unstructured":"Wu Y, Wang S, Song G, Huang Q (2019) Learning fragment self-attention embeddings for image-text retrieval. In: Proceedings of the 27th ACM international conference on multimedia, pp 2088\u20132096","DOI":"10.1145\/3343031.3350940"},{"key":"5823_CR38","doi-asserted-by":"crossref","unstructured":"Wu H, Liu Y, Cai H, He S (2022) Learning transferable perturbations for image captioning. ACM Transactions on Multimedia Computing, Communications, and Applications, pp 1\u201318","DOI":"10.1145\/3478024"},{"key":"5823_CR39","doi-asserted-by":"crossref","unstructured":"Yu Z, Yu J, Cui Y, Tao D, Tian Q (2019) Deep modular co-attention networks for visual question answering. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 6281\u20136290","DOI":"10.1109\/CVPR.2019.00644"},{"key":"5823_CR40","doi-asserted-by":"crossref","unstructured":"Yu R, Jin F, Qiao Z, Yuan Y, Wang G (2023) Multi-scale image-text matching network for scene and spatio-temporal images. Future Generation Computer Systems, pp 292\u2013300","DOI":"10.1016\/j.future.2023.01.004"},{"key":"5823_CR41","doi-asserted-by":"crossref","unstructured":"Zhang K, Mao Z, Wang Q, Zhang Y (2022) Negative-aware attention framework for image-text retrieval. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 15661\u201315670","DOI":"10.1109\/CVPR52688.2022.01521"},{"key":"5823_CR42","doi-asserted-by":"crossref","unstructured":"Zhang Q, Lei Z, Zhang Z, Li S (2020) Context-aware attention network for image-text retrieval. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 3536\u20133545","DOI":"10.1109\/CVPR42600.2020.00359"},{"key":"5823_CR43","doi-asserted-by":"crossref","unstructured":"Zhu J, Li Z, Zeng Y, Wei J, Ma H (2022) Image-text retrieval with fine-grained relational dependency and bidirectional attention-based generative networks. In: Proceedings of the 30th ACM international conference on multimedia, pp 395\u2013403","DOI":"10.1145\/3503161.3548058"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-024-05823-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-024-05823-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-024-05823-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,30]],"date-time":"2024-09-30T13:10:11Z","timestamp":1727701811000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-024-05823-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,9,11]]},"references-count":43,"journal-issue":{"issue":"23","published-print":{"date-parts":[[2024,12]]}},"alternative-id":["5823"],"URL":"https:\/\/doi.org\/10.1007\/s10489-024-05823-1","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,9,11]]},"assertion":[{"value":"28 August 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 September 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing Interests"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics Approval"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical Standards"}},{"value":"Not applicable.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to Participate"}},{"value":"Not applicable.","order":6,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for Publication"}},{"value":"Not applicable.","order":7,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical and Informed Consent for Data Used"}},{"value":"The authors declare that this manuscript is original, has not been published before, and is not currently being considered for publication elsewhere. The authors confirm that the manuscript has been read and approved by all named authors and that there are no other persons who satisfied the criteria for authorship but are not listed. The authors further confirm that the order of authors listed in the manuscript has been approved by all of us. The authors understand that the Corresponding Author is the sole contact for the Editorial process. He\/she is responsible for communicating with the other authors about progress, submissions of revisions, and the final approval of proofs.","order":8,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical Standards"}}]}}