{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:35:48Z","timestamp":1784421348203,"version":"3.55.0"},"reference-count":55,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2023,5,3]],"date-time":"2023-05-03T00:00:00Z","timestamp":1683072000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,5,3]],"date-time":"2023-05-03T00:00:00Z","timestamp":1683072000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Mach. Intell. Res."],"published-print":{"date-parts":[[2023,8]]},"DOI":"10.1007\/s11633-022-1386-4","type":"journal-article","created":{"date-parts":[[2023,5,2]],"date-time":"2023-05-02T23:02:11Z","timestamp":1683068531000},"page":"569-582","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":19,"title":["Cross-modal Contrastive Learning for Generalizable and Efficient Image-text Retrieval"],"prefix":"10.1007","volume":"20","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2620-6296","authenticated-orcid":false,"given":"Haoyu","family":"Lu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuqi","family":"Huo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mingyu","family":"Ding","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nanyi","family":"Fei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0280-7724","authenticated-orcid":false,"given":"Zhiwu","family":"Lu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,5,3]]},"reference":[{"key":"1386_CR1","doi-asserted-by":"publisher","first-page":"12652","DOI":"10.1109\/CVPR42600.2020.01267","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"H Chen","year":"2020","unstructured":"H. Chen, G. G. Ding, X. D. Liu, Z. J. Lin, J. Liu, J. G. Han. IMRAM: Iterative matching with recurrent attention memory for cross-modal image-text retrieval. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, IEEE, Seattle, USA, pp. 12652\u201312660, 2020. DOI: https:\/\/doi.org\/10.1109\/CVPR42600.2020.01267."},{"key":"1386_CR2","doi-asserted-by":"publisher","first-page":"212","DOI":"10.1007\/978-3-030-01225-0_13","volume-title":"Proceedings of the 15th European Conference on Computer Vision","author":"K H Lee","year":"2018","unstructured":"K. H. Lee, X. Chen, G. Hua, H. D. Hu, X. D. He. Stacked cross attention for image-text matching. In Proceedings of the 15th European Conference on Computer Vision, Springer, Munich, Germany, pp. 212\u2013228, 2018. DOI: https:\/\/doi.org\/10.1007\/978-3-030-01225-0_13."},{"key":"1386_CR3","unstructured":"H. Y. Lu, M. Y. Ding, N. Y. Fei, Y. Q. Huo, Z. W. Lu. LG-DN: Language-guided denoising network for video-language modeling. In Proceedings of Advances in Neural Information Processing Systems, 2022."},{"key":"1386_CR4","doi-asserted-by":"publisher","unstructured":"O. Vinyals, A. Toshev, S. Bengio, D. Erhan. Show and h]A neural image caption generator. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, Boston, USA, pp. 3156\u20133164, 2015. DOI: https:\/\/doi.org\/10.1109\/CVPR.2015.7298935.","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"1386_CR5","doi-asserted-by":"publisher","unstructured":"X. Jia, E. Gavves, B. Fernando, T. Tuytelaars. Guiding the long-short term memory model for image caption generation. In Proceedings of IEEE International Conference on Computer Vision, Santiago, Chile, pp. 2407\u20132415, 2015. DOI: https:\/\/doi.org\/10.1109\/ICCV.2015.277.","DOI":"10.1109\/ICCV.2015.277"},{"key":"1386_CR6","doi-asserted-by":"publisher","first-page":"1219","DOI":"10.1109\/CVPR.2018.00133","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"J Johnson","year":"2018","unstructured":"J. Johnson, A. Gupta, L. Fei-Fei. Image generation from scene graphs. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, IEEE, Salt Lake City, USA, pp. 1219\u20131228, 2018. DOI: https:\/\/doi.org\/10.1109\/CVPR.2018.00133."},{"key":"1386_CR7","doi-asserted-by":"publisher","unstructured":"T. T. Qiao, J. Zhang, D. Q. Xu, D. C. Tao. MirrorGAN: Learning text-to-image generation by redescription. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Long Beach, USA, pp. 1505\u20131514, 2019. DOI: https:\/\/doi.org\/10.1109\/CVPR.2019.00160.","DOI":"10.1109\/CVPR.2019.00160"},{"key":"1386_CR8","doi-asserted-by":"publisher","first-page":"3128","DOI":"10.1109\/CVPR.2015.7298932","volume-title":"Proceedings of IEEE Conference on Computer Vision and Pattern Recognition","author":"A Karpathy","year":"2015","unstructured":"A. Karpathy, F. F. Li. Deep visual-semantic alignments for generating image descriptions. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, IEEE, Boston, USA, pp. 3128\u20133137, 2015. DOI: https:\/\/doi.org\/10.1109\/CVPR.2015.7298932."},{"key":"1386_CR9","doi-asserted-by":"publisher","first-page":"104","DOI":"10.1007\/978-3-030-58577-8_7","volume-title":"Proceedings of the 16th European Conference on Computer Vision","author":"Y C Chen","year":"2020","unstructured":"Y. C. Chen, L. J. Li, L. C. Yu, A. El Kholy, F. Ahmed, Z. Gan, Y. Cheng, J. J. Liu. UNITER: UNiversal image-TExt representation learning. In Proceedings of the 16th European Conference on Computer Vision, Springer, Glasgow, UK, pp. 104\u2013120, 2020. DOI: https:\/\/doi.org\/10.1007\/978-3-030-58577-8_7."},{"key":"1386_CR10","unstructured":"R. Kiros, R. Salakhutdinov, R. S. Zemel. Unifying visual-semantic embeddings with multimodal neural language models. [Online], https:\/\/arxiv.org\/abs\/1411.2539, 2014."},{"key":"1386_CR11","doi-asserted-by":"publisher","unstructured":"L. W. Wang, Y. Li, S. Lazebnik. Learning deep structure-preserving image-text embeddings. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, Las Vegas, USA, pp. 5005\u20135013, 2016. DOI: https:\/\/doi.org\/10.1109\/CVPR.2016.541.","DOI":"10.1109\/CVPR.2016.541"},{"key":"1386_CR12","unstructured":"Y. Q. Huo, M. L. Zhang, G. Z. Liu, H. Y. Lu, Y. Z. Gao, G. X. Yang, J. Y. Wen, H. Zhang, B. G. Xu, W. H. Zheng, Z. Z. Xi, Y. Q. Yang, A. W. Hu, J. M. Zhao, R. C. Li, Y. D. Zhao, L. Zhang, Y. Q. Song, X. Hong, W. Q. Cui, D. Y. Hou, Y. Y. Li, J. Y. Li, P. Y. Liu, Z. Gong, C. H. Jin, Y. C. Sun, S. Z. Chen, Z. W. Lu, Z. C. Dou, Q. Jin, Y. Y. Lan, W. X. Zhao, R. H. Song, J. R. Wen. WenLan: Bridging vision and language by large-scale multi-modal pre-training. [Online], https:\/\/arxiv.org\/abs\/2103.06561, 2021."},{"key":"1386_CR13","doi-asserted-by":"publisher","unstructured":"N. Y. Fei, Z. W. Lu, Y. Z. Gao, G. X. Yang, Y. Q. Huo, J. Y. Wen, H. Y. Lu, R. H. Song, X. Gao, T. Xiang, H. Sun, J. R. Wen. Towards artificial general intelligence via a multimodal foundation model. Nature Communications, vol. 13, no. 1, Article number 3094, 2022. DOI: https:\/\/doi.org\/10.1038\/s41467-022-30761-2.","DOI":"10.1038\/s41467-022-30761-2"},{"key":"1386_CR14","doi-asserted-by":"publisher","first-page":"15671","DOI":"10.1109\/CVPR52688.2022.01524","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"H Y Lu","year":"2022","unstructured":"H. Y. Lu, N. Y. Fei, Y. Q. Huo, Y. Z. Gao, Z. W. Lu, J. R. Wen. COTS: Collaborative two-stream vision-language pre-training model for cross-modal retrieval. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, IEEE, New Orleans, USA, pp. 15671\u201315680, 2022. DOI: https:\/\/doi.org\/10.1109\/CVPR52688.2022.01524."},{"key":"1386_CR15","doi-asserted-by":"publisher","first-page":"2088","DOI":"10.1145\/3343031.3350940","volume-title":"Proceedings of the 27th ACM International Conference on Multimedia","author":"Y L Wu","year":"2019","unstructured":"Y. L. Wu, S. H. Wang, G. L. Song, Q. M. Huang. Learning fragment self-attention embeddings for image-text matching. In Proceedings of the 27th ACM International Conference on Multimedia, ACM, Nice, France, pp. 2088\u20132096, 2019. DOI: https:\/\/doi.org\/10.1145\/3343031.3350940."},{"issue":"2","key":"1386_CR16","doi-asserted-by":"publisher","first-page":"1218","DOI":"10.1609\/aaai.v35i2.16209","volume":"35","author":"H W Diao","year":"2021","unstructured":"H. W. Diao, Y. Zhang, L. Ma, H. C. Lu. Similarity reasoning and filtration for image-text matching. Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, no. 2, pp. 1218\u20131226, 2021. DOI: https:\/\/doi.org\/10.1609\/aaai.v35i2.16209.","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"1386_CR17","unstructured":"A. Dosovitskiy, L. Beyer, A. Kolesnikov, D. Weissenborn, X. H. Zhai, T. Unterthiner, M. Dehghani, M. Minderer, G. Heigold, S. Gelly, J. Uszkoreit, N. Houlsby. An image is worth 16\u00d716 words: Transformers for image recognition at scale. In Proceedings of the 9th International Conference on Learning Representations, 2021."},{"key":"1386_CR18","doi-asserted-by":"publisher","first-page":"3733","DOI":"10.1109\/CVPR.2018.00393","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Z R Wu","year":"2018","unstructured":"Z. R. Wu, Y. J. Xiong, S. X. Yu, D. H. Lin. Unsupervised feature learning via non-parametric instance discrimination. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, IEEE, Salt Lake City, USA, pp. 3733\u20133742, 2018. DOI: https:\/\/doi.org\/10.1109\/CVPR.2018.00393."},{"key":"1386_CR19","unstructured":"A. van den Oord, Y. Z. Li, O. Vinyals. Representation learning with contrastive predictive coding. [Online], https:\/\/arxiv.org\/abs\/1807.03748, 2018."},{"key":"1386_CR20","unstructured":"R. D. Hjelm, A. Fedorov, S. Lavoie-Marchildon, K. Grewal, P. Bachman, A. Trischler, Y. Bengio. Learning deep representations by mutual information estimation and maximization. In Proceedings of the 7th International Conference on Learning Representations, New Orleans, USA, 2019."},{"key":"1386_CR21","doi-asserted-by":"publisher","first-page":"6001","DOI":"10.1109\/ICCV.2019.00610","volume-title":"Proceedings of IEEE\/CVF International Conference on Computer Vision","author":"C X Zhuang","year":"2019","unstructured":"C. X. Zhuang, A. Zhai, D. Yamins. Local aggregation for unsupervised learning of visual embeddings. In Proceedings of IEEE\/CVF International Conference on Computer Vision, IEEE, Seoul, Republic of Korea, pp. 6001\u20136011, 2019. DOI: https:\/\/doi.org\/10.1109\/ICCV.2019.00610."},{"key":"1386_CR22","first-page":"15509","volume-title":"Proceedings of the 33rd Conference on Neural Information Processing Systems","author":"P Bachman","year":"2019","unstructured":"P. Bachman, R. D. Hjelm, W. Buchwalter. Learning representations by maximizing mutual information across views. In Proceedings of the 33rd Conference on Neural Information Processing Systems, ACM, Vancouver, Canada, pp. 15509\u201315519, 2019."},{"key":"1386_CR23","unstructured":"T. Chen, S. Kornblith, M. Norouzi, G. Hinton. A simple framework for contrastive learning of visual representations. In Proceedings of the 37th International Conference on Machine Learning, pp. 1597\u20131607, 2020."},{"key":"1386_CR24","first-page":"21271","volume-title":"Proceedings of the 34th Conference on Neural Information Processing Systems","author":"J B Grill","year":"2020","unstructured":"J. B. Grill, F. Strub, F. Altch\u00e9, C. Tallec, P. Richemond, E. Buchatskaya, C. Doersch, B. \u00c1. Pires, Z. H. Guo, M. G. Azar, B. Piot, K. Kavukcuoglu, R. Munos, M. Valko. Bootstrap your own latent-a new approach to self-supervised learning. In Proceedings of the 34th Conference on Neural Information Processing Systems, ACM, Vancouver, Canada, pp. 21271\u201321284, 2020."},{"key":"1386_CR25","doi-asserted-by":"publisher","first-page":"15750","DOI":"10.1109\/CVPR46437.2021.01549","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"X L Chen","year":"2021","unstructured":"X. L. Chen, K. M. He. Exploring simple Siamese representation learning. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, IEEE, Nashville, USA, pp. 15750\u201315758, 2021. DOI: https:\/\/doi.org\/10.1109\/CVPR46437.2021.01549."},{"issue":"4","key":"1386_CR26","doi-asserted-by":"publisher","first-page":"556","DOI":"10.1007\/s11633-021-1297-9","volume":"18","author":"D Y She","year":"2021","unstructured":"D. Y. She, K. Xu. Contrastive self-supervised representation learning using synthetic data. International Journal of Automation and Computing, vol. 18, no. 4, pp. 556\u2013567, 2021. DOI: https:\/\/doi.org\/10.1007\/s11633-021-1297-9.","journal-title":"International Journal of Automation and Computing"},{"key":"1386_CR27","first-page":"5998","volume-title":"Proceedings of the 31st Conference on Neural Information Processing Systems","author":"A Vaswani","year":"2017","unstructured":"A. Vaswani, N. Shazeer, N. Parmar, J. Uszkoreit, L. Jones, A. N. Gomez, L. Kaiser, I. Polosukhin. Attention is all you need. In Proceedings of the 31st Conference on Neural Information Processing Systems, ACM, Long Beach, USA, pp. 5998\u20136008, 2017."},{"key":"1386_CR28","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Proceedings of the 13th European Conference on Computer Vision","author":"T Y Lin","year":"2014","unstructured":"T. Y. Lin, M. Maire, S. Belongie, J. Hays, P. Perona, D. Ramanan, P. Doll\u00e1ar, C. L. Zitnick. Microsoft COCO: Common objects in context. In Proceedings of the 13th European Conference on Computer Vision, Springer, Zurich, Switzerland, pp. 740\u2013755, 2014. DOI: https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48."},{"issue":"1","key":"1386_CR29","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1162\/tacl_a_00166","volume":"2","author":"P Young","year":"2014","unstructured":"P. Young, A. Lai, M. Hodosh, J. Hockenmaier. From image descriptions to visual denotations: New similarity metrics for semantic inference over event descriptions. Transactions of the Association for Computational Linguistics, vol. 2, no. 1, pp. 67\u201378, 2014. DOI: https:\/\/doi.org\/10.1162\/tacl_a_00166.","journal-title":"Transactions of the Association for Computational Linguistics"},{"key":"1386_CR30","first-page":"91","volume-title":"Proceedings of the 28th International Conference on Neural Information Processing Systems","author":"S Q Ren","year":"2015","unstructured":"S. Q. Ren, K. M. He, R. Girshick, J. Sun. Faster R-CNN: Towards real-time object detection with region proposal networks. In Proceedings of the 28th International Conference on Neural Information Processing Systems, ACM, Montreal, Canada, pp. 91\u201399, 2015."},{"key":"1386_CR31","doi-asserted-by":"publisher","unstructured":"K. M. He, X. Y. Zhang, S. Q. Ren, J. Sun. Deep residual learning for image recognition. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, Las Vegas, USA, pp. 770\u2013778, 2016. DOI: https:\/\/doi.org\/10.1109\/CVPR.2016.90.","DOI":"10.1109\/CVPR.2016.90"},{"key":"1386_CR32","doi-asserted-by":"publisher","unstructured":"R. Girshick, J. Donahue, T. Darrell, J. Malik. Rich feature hierarchies for accurate object detection and semantic segmentation. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, Columbus, USA, pp. 580\u2013587, 2014. DOI: https:\/\/doi.org\/10.1109\/CVPR.2014.81.","DOI":"10.1109\/CVPR.2014.81"},{"key":"1386_CR33","doi-asserted-by":"publisher","first-page":"10938","DOI":"10.1109\/CVPR42600.2020.01095","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"X Wei","year":"2020","unstructured":"X. Wei, T. Z. Zhang, Y. Li, Y. D. Zhang, F. Wu. Multimodality cross attention network for image and sentence matching. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, IEEE, Seattle, USA, pp. 10938\u201310947, 2020. DOI: https:\/\/doi.org\/10.1109\/CVPR42600.2020.01095."},{"key":"1386_CR34","doi-asserted-by":"publisher","first-page":"6077","DOI":"10.1109\/CVPR.2018.00636","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"P Anderson","year":"2018","unstructured":"P. Anderson, X. D. He, C. Buehler, D. Teney, M. Johnson, S. Gould, L. Zhang. Bottom-up and top-down attention for image captioning and visual question answering. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, IEEE, Salt Lake City, USA, pp. 6077\u20136086, 2018. DOI: https:\/\/doi.org\/10.1109\/CVPR.2018.00636."},{"issue":"1","key":"1386_CR35","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","volume":"123","author":"R Krishna","year":"2017","unstructured":"R. Krishna, Y. K. Zhu, O. Groth, J. Johnson, K. Hata, J. Kravitz, S. Chen, Y. Kalantidis, L. J. Li, D. A. Shamma, M. S. Bernstein, L. Fei-Fei. Visual genome: Connecting language and vision using crowdsourced dense image annotations. International Journal of Computer Vision, vol. 123, no. 1, pp. 32\u201373, 2017. DOI: https:\/\/doi.org\/10.1007\/s11263-016-0981-7.","journal-title":"International Journal of Computer Vision"},{"key":"1386_CR36","doi-asserted-by":"publisher","first-page":"5763","DOI":"10.1109\/ICCV.2019.00586","volume-title":"Proceedings of IEEE\/CVF International Conference on Computer Vision","author":"Z H Wang","year":"2019","unstructured":"Z. H. Wang, X. H. Liu, H. S. Li, L. Sheng, J. J. Yan, X. G. Wang, J. Shao. CAMP: Cross-modal adaptive message passing for text-image retrieval. In Proceedings of IEEE\/CVF International Conference on Computer Vision, IEEE, Seoul, Republic of Korea, pp. 5763\u20135772, 2019. DOI: https:\/\/doi.org\/10.1109\/ICCV.2019.00586."},{"key":"1386_CR37","doi-asserted-by":"publisher","first-page":"707","DOI":"10.1007\/978-3-030-01246-5_42","volume-title":"Proceedings of the 15th European Conference on Computer Vision","author":"Y Zhang","year":"2018","unstructured":"Y. Zhang, H. C. Lu. Deep cross-modal projection learning for image-text matching. In Proceedings of the 15th European Conference on Computer Vision, Springer, Munich, Germany, pp. 707\u2013723, 2018. DOI: https:\/\/doi.org\/10.1007\/978-3-030-01246-5_42."},{"key":"1386_CR38","doi-asserted-by":"publisher","unstructured":"J. Devlin, M. W. Chang, K. Lee, K. Toutanova. BERT: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Minneapolis, USA, pp. 4171\u20134186, 2019. DOI: https:\/\/doi.org\/10.18653\/v1\/N19-1423.","DOI":"10.18653\/v1\/N19-1423"},{"key":"1386_CR39","doi-asserted-by":"publisher","first-page":"9726","DOI":"10.1109\/CVPR42600.2020.00975","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"K M He","year":"2020","unstructured":"K. M. He, H. Q. Fan, Y. X. Wu, S. N. Xie, R. Girshick. Momentum contrast for unsupervised visual representation learning. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, IEEE, Seattle, USA, pp. 9726\u20139735, 2020. DOI: https:\/\/doi.org\/10.1109\/CVPR42600.2020.00975."},{"key":"1386_CR40","first-page":"307","volume":"13","author":"M U Gutmann","year":"2012","unstructured":"M. U. Gutmann, A. Hyv\u00e4rinen. Noise-contrastive estimation of unnormalized statistical models, with applications to natural image statistics. Journal of Machine Learning Research, vol. 13, pp. 307\u2013361, 2012.","journal-title":"Journal of Machine Learning Research"},{"key":"1386_CR41","unstructured":"A. Radford, J. W. Kim, C. Hallacy, A. Ramesh, G. Goh, S. Agarwal, G. Sastry, A. Askell, P. Mishkin, J. Clark, G. Krueger, I. Sutskever. Learning transferable visual models from natural language supervision. In Proceedings of the 38th International Conference on Machine Learning, pp. 8748\u20138763, 2021."},{"key":"1386_CR42","unstructured":"Y. H. Liu, M. Ott, N. Goyal, J. F. Du, M. Joshi, D. Q. Chen, O. Levy, M. Lewis, L. Zettlemoyer, V. Stoyanov. RoBERTa: A robustly optimized BERT pretraining approach. [Online], https:\/\/arxiv.org\/abs\/1907.11692, 2019."},{"key":"1386_CR43","first-page":"807","volume-title":"Proceedings of the 27th International Conference on Machine Learning","author":"V Nair","year":"2010","unstructured":"V. Nair, G. E. Hinton. Rectified linear units improve restricted boltzmann machines. In Proceedings of the 27th International Conference on Machine Learning, Omnipress, Haifa, Israel, pp. 807\u2013814, 2010."},{"key":"1386_CR44","first-page":"2121","volume-title":"Proceedings of the 26th International Conference on Neural Information Processing Systems","author":"A Frome","year":"2013","unstructured":"A. Frome, G. S. Corrado, J. Shlens, S. Bengio, J. Dean, M. Ranzato, T. Mikolov. DeViSE: A deep visual-semantic embedding model. In Proceedings of the 26th International Conference on Neural Information Processing Systems, ACM, Lake Tahoe, USA, pp. 2121\u20132129, 2013."},{"key":"1386_CR45","unstructured":"M. X. Tan, Q. Le. EfficientNet: Rethinking model scaling for convolutional neural networks. In Proceedings of the 36th International Conference on Machine Learning, Long Beach, USA, pp. 6105\u20136114, 2019."},{"key":"1386_CR46","doi-asserted-by":"publisher","first-page":"3533","DOI":"10.1109\/CVPR42600.2020.00359","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Q Zhang","year":"2020","unstructured":"Q. Zhang, Z. Lei, Z. X. Zhang, S. Z. Li. Context-aware attention network for image-text retrieval. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, IEEE, Seattle, USA, pp. 3533\u20133542, 2020. DOI: https:\/\/doi.org\/10.1109\/CVPR42600.2020.00359."},{"key":"1386_CR47","doi-asserted-by":"publisher","first-page":"15789","DOI":"10.1109\/CVPR46437.2021.01553","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"J C Chen","year":"2021","unstructured":"J. C. Chen, H. X. Hu, H. Wu, Y. N. Jiang, C. H. Wang. Learning the best pooling strategy for visual semantic embedding. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, IEEE, Nashville, USA, pp. 15789\u201315798, 2021. DOI: https:\/\/doi.org\/10.1109\/CVPR46437.2021.01553."},{"key":"1386_CR48","unstructured":"W. Kim, B. Son, I. Kim. ViLT: Vision-and-language transformer without convolution or region supervision. In Proceedings of the 38th International Conference on Machine Learning, pp. 5583\u20135594, 2021."},{"key":"1386_CR49","doi-asserted-by":"publisher","first-page":"18145","DOI":"10.1109\/CVPR52688.2022.01763","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Z Y Dou","year":"2022","unstructured":"Z. Y. Dou, Y. C. Xu, Z. Gan, J. F. Wang, S. H. Wang, L. J. Wang, C. G. Zhu, P. C. Zhang, L. Yuan, N. Y. Peng, Z. C. Liu, M. Zeng. An empirical study of training end-to-end vision-and-language transformers. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, IEEE, New Orleans, USA, pp. 18145\u201318155, 2022. DOI: https:\/\/doi.org\/10.1109\/CVPR52688.2022.01763."},{"key":"1386_CR50","doi-asserted-by":"publisher","first-page":"121","DOI":"10.1007\/978-3-030-58577-8_8","volume-title":"Proceedings of the 16th European Conference on Computer Vision","author":"X J Li","year":"2020","unstructured":"X. J. Li, X. Yin, C. Y. Li, P. C. Zhang, X. W. Hu, L. Zhang, L. J. Wang, H. D. Hu, L. Dong, F. R. Wei, Y. J. Choi, J. F. Gao. OSCAR: Object-semantics aligned pretraining for vision-language tasks. In Proceedings of the 16th European Conference on Computer Vision, Springer, Glasgow, UK, pp. 121\u2013137, 2020. DOI: https:\/\/doi.org\/10.1007\/978-3-030-58577-8_8."},{"key":"1386_CR51","doi-asserted-by":"publisher","first-page":"5753","DOI":"10.1109\/ICCV.2019.00585","volume-title":"Proceedings of IEEE\/CVF International Conference on Computer Vision","author":"Z Ji","year":"2019","unstructured":"Z. Ji, H. R. Wang, J. G. Han, Y. W. Pang. Saliency-guided attention network for image-sentence matching. In Proceedings of IEEE\/CVF International Conference on Computer Vision, IEEE, Seoul, Republic of Korea, pp. 5753\u20135762, 2019. DOI: https:\/\/doi.org\/10.1109\/ICCV.2019.00585."},{"key":"1386_CR52","doi-asserted-by":"publisher","unstructured":"W. Li, C. Gao, G. C. Niu, X. Y. Xiao, H. Liu, J. C. Liu, H. Wu, H. F. Wang. UNIMO: Towards unified-modal understanding and generation via cross-modal contrastive learning. In Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing, pp. 2592\u20132607, 2021. DOI: https:\/\/doi.org\/10.18653\/v1\/2021.acl-long.202.","DOI":"10.18653\/v1\/2021.acl-long.202"},{"key":"1386_CR53","doi-asserted-by":"publisher","unstructured":"Y. X. Wang, H. Yang, X. M. Qian, L. Ma, J. Lu, B. Li, X. Fan. Position focused attention network for image-text matching. In Proceedings of the 28th International Joint Conference on Artificial Intelligence, Macao, China, pp. 3792\u20133798, 2019. DOI: https:\/\/doi.org\/10.24963\/ijcai.2019\/526.","DOI":"10.24963\/ijcai.2019\/526"},{"key":"1386_CR54","doi-asserted-by":"publisher","unstructured":"F. Yan, K. Mikolajczyk. Deep correlation for matching images and text. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, Boston, USA, pp. 3441\u20133450, 2015. DOI: https:\/\/doi.org\/10.1109\/CVPR.2015.7298966.","DOI":"10.1109\/CVPR.2015.7298966"},{"key":"1386_CR55","doi-asserted-by":"publisher","first-page":"1979","DOI":"10.1109\/CVPR.2019.00208","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Y L Song","year":"2019","unstructured":"Y. L. Song, M. Soleymani. Polysemous visual-semantic embedding for cross-modal retrieval. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, IEEE, Long Beach, USA, pp. 1979\u20131988, 2019. DOI: https:\/\/doi.org\/10.1109\/CVPR.2019.00208."}],"container-title":["Machine Intelligence Research"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11633-022-1386-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11633-022-1386-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11633-022-1386-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,7,15]],"date-time":"2023-07-15T13:06:49Z","timestamp":1689426409000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11633-022-1386-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,5,3]]},"references-count":55,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2023,8]]}},"alternative-id":["1386"],"URL":"https:\/\/doi.org\/10.1007\/s11633-022-1386-4","relation":{},"ISSN":["2731-538X","2731-5398"],"issn-type":[{"value":"2731-538X","type":"print"},{"value":"2731-5398","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,5,3]]},"assertion":[{"value":"24 May 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 October 2022","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 May 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}