{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,28]],"date-time":"2026-01-28T05:23:00Z","timestamp":1769577780982,"version":"3.49.0"},"reference-count":46,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"10","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2025,10,1]]},"DOI":"10.1587\/transinf.2024edp7279","type":"journal-article","created":{"date-parts":[[2025,4,16]],"date-time":"2025-04-16T18:12:36Z","timestamp":1744827156000},"page":"1230-1238","source":"Crossref","is-referenced-by-count":1,"title":["Cross-Modal Deep Interaction and Semantic Aligning for Image-Text Retrieval"],"prefix":"10.1587","volume":"E108.D","author":[{"given":"Ruidong","family":"CHEN","sequence":"first","affiliation":[{"name":"Guangxi Key Laboratory of Image and Graphic Intelligent Processing, Guilin University of Electronic Technology"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Baohua","family":"QIANG","sequence":"additional","affiliation":[{"name":"Guangxi Key Laboratory of Image and Graphic Intelligent Processing, Guilin University of Electronic Technology"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xianyi","family":"YANG","sequence":"additional","affiliation":[{"name":"Guangxi Key Laboratory of Image and Graphic Intelligent Processing, Guilin University of Electronic Technology"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shihao","family":"ZHANG","sequence":"additional","affiliation":[{"name":"Guangxi Key Laboratory of Image and Graphic Intelligent Processing, Guilin University of Electronic Technology"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuan","family":"XIE","sequence":"additional","affiliation":[{"name":"Guangxi Key Laboratory of Image and Graphic Intelligent Processing, Guilin University of Electronic Technology"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"key":"1","doi-asserted-by":"publisher","unstructured":"[1] F. Wu, S. Li, G. Peng, Y. Ma, and X.-Y. Jing, \u201cModality-fused graph network for cross-modal retrieval,\u201d IEICE Trans. Inf. &amp; Syst., vol.E106-D, no.5, pp.1094-1097, 2023. 10.1587\/transinf.2022edl8069","DOI":"10.1587\/transinf.2022EDL8069"},{"key":"2","doi-asserted-by":"crossref","unstructured":"[2] F. Feng, X. Wang, and R. Li, \u201cCross-modal retrieval with correspondence autoencoder,\u201d Proc. 22nd ACM international conference on Multimedia, pp.7-16, 2014. 10.1145\/2647868.2654902","DOI":"10.1145\/2647868.2654902"},{"key":"3","unstructured":"[3] Y. Peng, X. Huang, and J. Qi, \u201cCross-media shared representation by hierarchical learning with multiple deep networks,\u201d Proc. Twenty-Fifth International Joint Conference on Artificial Intelligence, vol.3846, p.3853, 2016."},{"key":"4","doi-asserted-by":"publisher","unstructured":"[4] Y. Peng, J. Qi, X. Huang, and Y. Yuan, \u201cCCL: Cross-modal correlation learning with multigrained fusion by hierarchical network,\u201d IEEE Trans. Multimedia, vol.20, no.2, pp.405-420, 2017. 10.1109\/tmm.2017.2742704","DOI":"10.1109\/TMM.2017.2742704"},{"key":"5","unstructured":"[5] J. Ngiam, A. Khosla, M. Kim, J. Nam, H. Lee, and A.Y. Ng, \u201cMultimodal deep learning,\u201d Proc. 28th International Conference on Machine Learning, pp.689-696, 2011."},{"key":"6","unstructured":"[6] N. Srivastava and R. Salakhutdinov, \u201cLearning representations for multimodal data with deep belief nets,\u201d International Conference on Machine Learning Workshop, vol.79, pp.978-1, 2012."},{"key":"7","unstructured":"[7] N. Srivastava and R.R. Salakhutdinov, \u201cMultimodal learning with deep Boltzmann machines,\u201d Advances in Neural Information Processing Systems, vol.25, 2012."},{"key":"8","doi-asserted-by":"crossref","unstructured":"[8] Z. Zeng, Y. Sun, and W. Mao, \u201cMCCN: Multimodal coordinated clustering network for large-scale cross-modal retrieval,\u201d Proc. 29th ACM International Conference on Multimedia, pp.5427-5435, 2021. 10.1145\/3474085.3475670","DOI":"10.1145\/3474085.3475670"},{"key":"9","doi-asserted-by":"crossref","unstructured":"[9] L. Zhen, P. Hu, X. Wang, and D. Peng, \u201cDeep supervised cross-modal retrieval,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.10394-10403, 2019. 10.1109\/cvpr.2019.01064","DOI":"10.1109\/CVPR.2019.01064"},{"key":"10","doi-asserted-by":"crossref","unstructured":"[10] Z. Zeng, S. Wang, N. Xu, and W. Mao, \u201cPAN: Prototype-based adaptive network for robust cross-modal retrieval,\u201d Proc. 44th International ACM SIGIR Conference on Research and Development in Information Retrieval, pp.1125-1134, 2021. 10.1145\/3404835.3462867","DOI":"10.1145\/3404835.3462867"},{"key":"11","doi-asserted-by":"publisher","unstructured":"[11] Y. Peng and J. Qi, \u201cCM-GANs: Cross-modal generative adversarial networks for common representation learning,\u201d ACM Transactions on Multimedia Computing, Communications, and Applications, vol.15, no.1, pp.1-24, 2019. 10.1145\/3284750","DOI":"10.1145\/3284750"},{"key":"12","doi-asserted-by":"crossref","unstructured":"[12] B. Wang, Y. Yang, X. Xu, A. Hanjalic, and H.T. Shen, \u201cAdversarial cross-modal retrieval,\u201d Proc. 25th ACM international conference on Multimedia, pp.154-162, 2017. 10.1145\/3123266.3123326","DOI":"10.1145\/3123266.3123326"},{"key":"13","doi-asserted-by":"crossref","unstructured":"[13] D. Kim, N. Kim, and S. Kwak, \u201cImproving cross-modal retrieval with set of diverse embeddings,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.23422-23431, 2023.","DOI":"10.1109\/CVPR52729.2023.02243"},{"key":"14","doi-asserted-by":"publisher","unstructured":"[14] L. Zhao, Y. Wang, J. Kato, and Y. Ishikawa, \u201cLearning local similarity with spatial interrelations on content-based image retrieval,\u201d IEICE Trans. Inf. &amp; Syst., vol.E106-D, no.5, pp.1069-1080, 2023. 10.1587\/transinf.2022edp7163","DOI":"10.1587\/transinf.2022EDP7163"},{"key":"15","doi-asserted-by":"publisher","unstructured":"[15] K. Sun and J. Zhu, \u201cSearching and learning discriminative regions for fine-grained image retrieval and classification,\u201d IEICE Trans. Inf. &amp; Syst., vol.E105-D, no.1, pp.141-149, 2022. 10.1587\/transinf.2021edp7094","DOI":"10.1587\/transinf.2021EDP7094"},{"key":"16","doi-asserted-by":"publisher","unstructured":"[16] J. Luo, C. He, and H. Luo, \u201cBRsyn-Caps: Chinese text classification using capsule network based on bert and dependency syntax,\u201d IEICE Trans. Inf. &amp; Syst., vol.E107-D, no.2, pp.212-219, 2024. 10.1587\/transinf.2023edp7119","DOI":"10.1587\/transinf.2023EDP7119"},{"key":"17","doi-asserted-by":"publisher","unstructured":"[17] Y. Tsuchida, K. Kubo, and H. Koga, \u201cContinuous similarity search for dynamic text streams,\u201d IEICE Trans. Inf. &amp; Syst., vol.E106-D, no.12, pp.2026-2035, 2023. 10.1587\/transinf.2022edp7229","DOI":"10.1587\/transinf.2022EDP7229"},{"key":"18","unstructured":"[18] A. Radford, J.W. Kim, C. Hallacy, A. Ramesh, G. Goh, S. Agarwal, G. Sastry, A. Askell, P. Mishkin, J. Clark, G. Krueger, and I. Sutskever, \u201cLearning transferable visual models from natural language supervision,\u201d International Conference on Machine Learning, pp.8748-8763, 2021."},{"key":"19","doi-asserted-by":"crossref","unstructured":"[19] N. Rasiwasia, J.C. Pereira, E. Coviello, G. Doyle, G.R.G. Lanckriet, R. Levy, and N. Vasconcelos, \u201cA new approach to cross-modal multimedia retrieval,\u201d Proc. 18th ACM international conference on Multimedia, pp.251-260, 2010. 10.1145\/1873951.1873987","DOI":"10.1145\/1873951.1873987"},{"key":"20","unstructured":"[20] C. Rashtchian, P. Young, M. Hodosh, and J. Hockenmaier, \u201cCollecting image annotations using amazon\u2019s mechanical turk,\u201d Proc. NAACL HLT 2010 Workshop on Creating Speech and Language Data with Amazon\u2019s Mechanical Turk, pp.139-147, 2010."},{"key":"21","doi-asserted-by":"crossref","unstructured":"[21] T.-S. Chua, J. Tang, R. Hong, H. Li, Z. Luo, and Y. Zheng, \u201cNus-wide: a real-world web image database from national university of singapore,\u201d Proc. ACM international conference on image and video retrieval, pp.1-9, 2009. 10.1145\/1646396.1646452","DOI":"10.1145\/1646396.1646452"},{"key":"22","doi-asserted-by":"publisher","unstructured":"[22] P. Hu, Z. Huang, D. Peng, X. Wang, and X. Peng, \u201cCross-modal retrieval with partially mismatched pairs,\u201d IEEE Trans. Pattern Anal. Mach. Intell., vol.45, no.8, pp.9595-9610, 2023. 10.1109\/tpami.2023.3247939","DOI":"10.1109\/TPAMI.2023.3247939"},{"key":"23","doi-asserted-by":"publisher","unstructured":"[23] Z. Liu, F. Zhao, and M. Zhang, \u201cAn efficient multimodal aggregation network for video-text retrieval,\u201d IEICE Trans. Inf. &amp; Syst., vol.E105-D, no.10, pp.1825-1828, 2022. 10.1587\/transinf.2022edl8018","DOI":"10.1587\/transinf.2022EDL8018"},{"key":"24","doi-asserted-by":"publisher","unstructured":"[24] F. Feng, R. Li, and X. Wang, \u201cDeep correspondence restricted boltzmann machine for cross-modal retrieval,\u201d Neurocomputing, vol.154, pp.50-60, 2015. 10.1016\/j.neucom.2014.12.020","DOI":"10.1016\/j.neucom.2014.12.020"},{"key":"25","doi-asserted-by":"publisher","unstructured":"[25] F. Kou, J. Du, W. Cui, L. Shi, P. Cheng, J. Chen, and J. Li, \u201cCommon semantic representation method based on object attention and adversarial learning for cross-modal data in iov,\u201d IEEE Trans. Veh. Technol., vol.68, no.12, pp.11588-11598, 2019. 10.1109\/tvt.2018.2890405","DOI":"10.1109\/TVT.2018.2890405"},{"key":"26","doi-asserted-by":"publisher","unstructured":"[26] L. Shi, J. Du, G. Cheng, X. Liu, Z. Xiong, and J. Luo, \u201cCross-media search method based on complementary attention and generative adversarial network for social networks,\u201d International Journal of Intelligent Systems, vol.37, no.8, pp.4393-4416, 2022. 10.1002\/int.22723","DOI":"10.1002\/int.22723"},{"key":"27","doi-asserted-by":"publisher","unstructured":"[27] X. Xu, K. Lin, Y. Yang, A. Hanjalic, and H.T. Shen, \u201cJoint feature synthesis and embedding: Adversarial cross-modal retrieval revisited,\u201d IEEE Trans. Pattern Anal. Mach. Intell., vol.44, no.6, pp.3030-3047, 2020. 10.1109\/tpami.2020.3045530","DOI":"10.1109\/TPAMI.2020.3045530"},{"key":"28","doi-asserted-by":"crossref","unstructured":"[28] C. Li, C. Deng, N. Li, W. Liu, X. Gao, and D. Tao, \u201cSelf-supervised adversarial hashing networks for cross-modal retrieval,\u201d Proc. IEEE conference on computer vision and pattern recognition, pp.4242-4251, 2018. 10.1109\/cvpr.2018.00446","DOI":"10.1109\/CVPR.2018.00446"},{"key":"29","doi-asserted-by":"publisher","unstructured":"[29] X. Huang, Y. Peng, and M. Yuan, \u201cMHTN: Modal-adversarial hybrid transfer network for cross-modal retrieval,\u201d IEEE Trans. Cybern., vol.50, no.3, pp.1047-1059, 2018. 10.1109\/tcyb.2018.2879846","DOI":"10.1109\/TCYB.2018.2879846"},{"key":"30","doi-asserted-by":"publisher","unstructured":"[30] L. Zhen, P. Hu, X. Peng, R.S.M. Goh, and J.T. Zhou, \u201cDeep multimodal transfer learning for cross-modal retrieval,\u201d IEEE Trans. Neural Netw. Learn. Syst., vol.33, no.2, pp.798-810, 2020. 10.1109\/tnnls.2020.3029181","DOI":"10.1109\/TNNLS.2020.3029181"},{"key":"31","doi-asserted-by":"publisher","unstructured":"[31] K. Zhao, P. Song, S. Li, W. Zhang, and W. Zheng, \u201cA novel adaptive weighted transfer subspace learning method for cross-database speech emotion recognition,\u201d IEICE Trans. Inf. &amp; Syst., vol.E105-D, no.9, pp.1643-1646, 2022. 10.1587\/transinf.2022edl8021","DOI":"10.1587\/transinf.2022EDL8021"},{"key":"32","doi-asserted-by":"publisher","unstructured":"[32] Z. Zeng, S. He, Y. Zhang, and W. Mao, \u201cA multimodal embedding transfer approach for consistent and selective learning processes in cross-modal retrieval,\u201d Information Sciences, vol.704, no.2025, p.121974, 2025. 10.1016\/j.ins.2025.121974","DOI":"10.1016\/j.ins.2025.121974"},{"key":"33","doi-asserted-by":"publisher","unstructured":"[33] Y. Zeng, X. Zhang, H. Li, J. Wang, J. Zhang, and W. Zhou, \u201cX<sup>2<\/sup>-VLM: All-in-one pre-trained model for vision-language tasks,\u201d IEEE Trans. Pattern Anal. Mach. Intell., vol.46, no.5, pp.3156-3168, 2023. 10.1109\/tpami.2023.3339661","DOI":"10.1109\/TPAMI.2023.3339661"},{"key":"34","doi-asserted-by":"crossref","unstructured":"[34] Y. Zhang, Y. Pan, T. Yao, R. Huang, T. Mei, and C.-W. Chen, \u201cLearning to generate language-supervised and open-vocabulary scene graph using pre-trained visual-semantic space,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.2915-2924, 2023. 10.1109\/cvpr52729.2023.00285","DOI":"10.1109\/CVPR52729.2023.00285"},{"key":"35","doi-asserted-by":"crossref","unstructured":"[35] A. Yang, A. Nagrani, P.H. Seo, A. Miech, J. Pont-Tuset, I. Laptev, J. Sivic, and C. Schmid, \u201cVid2seq: Large-scale pretraining of a visual language model for dense video captioning,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.10714-10726, 2023. 10.1109\/cvpr52729.2023.01032","DOI":"10.1109\/CVPR52729.2023.01032"},{"key":"36","doi-asserted-by":"crossref","unstructured":"[36] Z. Hu, A. Iscen, C. Sun, Z. Wang, K.-W. Chang, Y. Sun, C. Schmid, D.A. Ross, and A. Fathi, \u201cReveal: Retrieval-augmented visual-language pre-training with multi-source multimodal knowledge memory,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.23369-23379, 2023. 10.1109\/cvpr52729.2023.02238","DOI":"10.1109\/CVPR52729.2023.02238"},{"key":"37","unstructured":"[37] Z. Zeng and W. Mao, \u201cA comprehensive empirical study of vision-language pre-trained model for supervised cross-modal retrieval,\u201d arXiv preprint arXiv:2201.02772, 2022."},{"key":"38","doi-asserted-by":"publisher","unstructured":"[38] D. Shi, L. Zhu, J. Li, G. Dong, and H. Zhang, \u201cIncomplete cross-modal retrieval with deep correlation transfer,\u201d ACM Transactions on Multimedia Computing, Communications and Applications, vol.20, no.5, pp.1-21, 2024. 10.1145\/3637442","DOI":"10.1145\/3637442"},{"key":"39","doi-asserted-by":"crossref","unstructured":"[39] J.M. Kim, A Koepke, C. Schmid, and Z. Akata, \u201cExposing and mitigating spurious correlations for cross-modal retrieval,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.2584-2594, 2023. 10.1109\/cvprw59228.2023.00257","DOI":"10.1109\/CVPRW59228.2023.00257"},{"key":"40","doi-asserted-by":"crossref","unstructured":"[40] K. He, X. Zhang, S. Ren, and J. Sun, \u201cDeep residual learning for image recognition,\u201d Proc. IEEE conference on computer vision and pattern recognition, pp.770-778, 2016. 10.1109\/cvpr.2016.90","DOI":"10.1109\/CVPR.2016.90"},{"key":"41","unstructured":"[41] A. Vaswani, N. Shazeer, N. Parmar, J. Uszkoreit, L. Jones, A.N. Gomez, L. Kaiser, and I. Polosukhin, \u201cAttention is all you need,\u201d Advances in Neural Information Processing Systems, vol.30, 2017."},{"key":"42","doi-asserted-by":"crossref","unstructured":"[42] J. Deng, J. Guo, N. Xue, and S. Zafeiriou, \u201cArcface: Additive angular margin loss for deep face recognition,\u201d Proc. IEEE\/CVF conference on computer vision and pattern recognition, pp.4690-4699, 2019. 10.1109\/cvpr.2019.00482","DOI":"10.1109\/CVPR.2019.00482"},{"key":"43","unstructured":"[43] D.P. Kingma and J. Ba, \u201cAdam: A method for stochastic optimization,\u201d arXiv preprint arXiv:1412.6980, 2014."},{"key":"44","unstructured":"[44] A. van den Oord, Y. Li, and O. Vinyals, \u201cRepresentation learning with contrastive predictive coding,\u201d arXiv preprint arXiv:1807.03748, 2018."},{"key":"45","doi-asserted-by":"crossref","unstructured":"[45] D. Jiang and M. Ye, \u201cCross-modal implicit relation reasoning and aligning for text-to-image person retrieval\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.2787-2797, 2023. 10.1109\/cvpr52729.2023.00273","DOI":"10.1109\/CVPR52729.2023.00273"},{"key":"46","doi-asserted-by":"crossref","unstructured":"[46] Z. Wang, Z. Gao, K. Guo, Y. Yang, X. Wang, and H.T. Shen, \u201cMultilateral semantic relations modeling for image text retrieval,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.2830-2839, 2023. 10.1109\/cvpr52729.2023.00277","DOI":"10.1109\/CVPR52729.2023.00277"}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E108.D\/10\/E108.D_2024EDP7279\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,4]],"date-time":"2025-10-04T03:28:47Z","timestamp":1759548527000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E108.D\/10\/E108.D_2024EDP7279\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,1]]},"references-count":46,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2025]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2024edp7279","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"value":"0916-8532","type":"print"},{"value":"1745-1361","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,10,1]]},"article-number":"2024EDP7279"}}