{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T03:14:32Z","timestamp":1778037272004,"version":"3.51.4"},"reference-count":42,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2026,1,7]],"date-time":"2026-01-07T00:00:00Z","timestamp":1767744000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,7]],"date-time":"2026-01-07T00:00:00Z","timestamp":1767744000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Innovation Project of GUET Graduate Education","award":["2024YCXB08"],"award-info":[{"award-number":["2024YCXB08"]}]},{"name":"Innovation Project of GUET Graduate Education","award":["2024YCXS037"],"award-info":[{"award-number":["2024YCXS037"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62262006"],"award-info":[{"award-number":["62262006"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100017691","name":"Guangxi Key Research and Development Program","doi-asserted-by":"publisher","award":["AB24010112"],"award-info":[{"award-number":["AB24010112"]}],"id":[{"id":"10.13039\/501100017691","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100017691","name":"Guangxi Key Research and Development Program","doi-asserted-by":"publisher","award":["AB23026048"],"award-info":[{"award-number":["AB23026048"]}],"id":[{"id":"10.13039\/501100017691","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Guilin Science and Technology Development Program","award":["20230110-1"],"award-info":[{"award-number":["20230110-1"]}]},{"name":"Natural Science Foundation of Guangxi","award":["2019GXNSFDA185007"],"award-info":[{"award-number":["2019GXNSFDA185007"]}]},{"name":"Natural Science Foundation of Guangxi","award":["2019GXNSFDA185006"],"award-info":[{"award-number":["2019GXNSFDA185006"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Intell Inf Syst"],"published-print":{"date-parts":[[2026,4]]},"DOI":"10.1007\/s10844-025-01019-2","type":"journal-article","created":{"date-parts":[[2026,1,7]],"date-time":"2026-01-07T07:33:16Z","timestamp":1767771196000},"page":"715-734","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Cross-modal quaternion relation mining and gap bridging for image-text retrieval"],"prefix":"10.1007","volume":"64","author":[{"given":"Ruidong","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Baohua","family":"Qiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xianyi","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shihao","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuan","family":"Xie","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lirui","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,1,7]]},"reference":[{"key":"1019_CR1","doi-asserted-by":"publisher","unstructured":"Cao, M., Li, S., & Li, J., et al. (2022). Image-text retrieval: A survey on recent research and development. arXiv preprint arXiv:2203.14713. https:\/\/doi.org\/10.48550\/arXiv.2203.14713","DOI":"10.48550\/arXiv.2203.14713"},{"key":"1019_CR2","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/TGRS.2025.3587097","volume":"63","author":"X Chen","year":"2025","unstructured":"Chen, X., Zheng, X., & Lu, X. (2025). Relevance-guided adaptive learning for remote sensing image-text retrieval. IEEE Transactions on Geoscience and Remote Sensing.,63, 1\u201313. https:\/\/doi.org\/10.1109\/TGRS.2025.3587097","journal-title":"IEEE Transactions on Geoscience and Remote Sensing."},{"key":"1019_CR3","doi-asserted-by":"publisher","unstructured":"Chua, T.-S., Tang, J., & Hong, R., et al. (2009). Nus-wide: a real-world web image database from national university of singapore. In: Proceedings of the ACM International Conference on Image and Video Retrieval, pp. 1\u20139. https:\/\/doi.org\/10.1145\/1646396.1646452","DOI":"10.1145\/1646396.1646452"},{"key":"1019_CR4","doi-asserted-by":"publisher","unstructured":"Diederik, P.K., & Jimmy, B. (2015). Adam: A method for stochastic optimization. In: 3rd International Conference on Learning Representations, pp. 1\u201315. https:\/\/doi.org\/10.48550\/arXiv.1412.6980","DOI":"10.48550\/arXiv.1412.6980"},{"key":"1019_CR5","doi-asserted-by":"publisher","unstructured":"El\u00a0Zaar, A., Mansouri, A., & Benaya, N., et al. (2025). Hybrid transformer-cnn architecture for multivariate time series forecasting: Integrating attention mechanisms with convolutional feature extraction. Journal of Intelligent Information Systems, pp. 1\u201332. https:\/\/doi.org\/10.1007\/s10844-025-00937-5","DOI":"10.1007\/s10844-025-00937-5"},{"key":"1019_CR6","doi-asserted-by":"publisher","unstructured":"Gu, S., Clark, C., & Kembhavi, A. (2023). I can\u2019t believe there\u2019s no images! learning visual tasks using only language supervision. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2672\u20132683. https:\/\/doi.org\/10.1109\/ICCV51070.2023.00252","DOI":"10.1109\/ICCV51070.2023.00252"},{"issue":"12","key":"1019_CR7","doi-asserted-by":"publisher","first-page":"2639","DOI":"10.1162\/0899766042321814","volume":"16","author":"DR Hardoon","year":"2004","unstructured":"Hardoon, D. R., Szedmak, S., & Shawe-Taylor, J. (2004). Canonical correlation analysis: An overview with application to learning methods. Neural Computation.,16(12), 2639\u20132664. https:\/\/doi.org\/10.1162\/0899766042321814","journal-title":"Neural Computation."},{"key":"1019_CR8","doi-asserted-by":"publisher","unstructured":"He, T., Zhang, Z., & Zhang, H., et al. (2019). Bag of tricks for image classification with convolutional neural networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 558\u2013567. https:\/\/doi.org\/10.1109\/CVPR.2019.00065","DOI":"10.1109\/CVPR.2019.00065"},{"key":"1019_CR9","doi-asserted-by":"publisher","unstructured":"Huang, J., Kang, P., & Fang, X., et al. (2024). Efficient discriminative hashing for cross-modal retrieval. IEEE Transactions on Systems, Man, and Cybernetics: Systems.,\u00a054(6), 3865\u20133878. https:\/\/doi.org\/10.1109\/TSMC.2024.3373612","DOI":"10.1109\/TSMC.2024.3373612"},{"key":"1019_CR10","doi-asserted-by":"publisher","unstructured":"Jiang, Q.-Y., & Li, W.-J. (2017). Deep cross-modal hashing. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3232\u20133240. https:\/\/doi.org\/10.1109\/CVPR.2017.348","DOI":"10.1109\/CVPR.2017.348"},{"key":"1019_CR11","doi-asserted-by":"publisher","unstructured":"Kang, C., Xiang, S., & Liao, S., et al. (2015). Learning consistent feature representation for cross-modal multimedia retrieval. IEEE Transactions on Multimedia.,\u00a017(3), 370\u2013381. https:\/\/doi.org\/10.1109\/TMM.2015.2390499","DOI":"10.1109\/TMM.2015.2390499"},{"issue":"3","key":"1019_CR12","doi-asserted-by":"publisher","first-page":"673","DOI":"10.1007\/s10844-023-00793-1","volume":"60","author":"NAA Khleel","year":"2023","unstructured":"Khleel, N. A. A., & Neh\u00e9z, K. (2023). A novel approach for software defect prediction using cnn and gru based on smote tomek method. Journal of Intelligent Information Systems.,60(3), 673\u2013707. https:\/\/doi.org\/10.1007\/s10844-023-00793-1","journal-title":"Journal of Intelligent Information Systems."},{"key":"1019_CR13","doi-asserted-by":"publisher","unstructured":"Li, D., Dimitrova, N., & Li, M., et al. (2003). Multimedia content processing through cross-modal association. In: Proceedings of the Eleventh ACM International Conference on Multimedia, pp. 604\u2013611. https:\/\/doi.org\/10.1145\/957013.957143","DOI":"10.1145\/957013.957143"},{"key":"1019_CR14","doi-asserted-by":"publisher","unstructured":"Liang, V. W., Zhang, Y., & Kwon, Y., et al. (2022). Mind the gap: Understanding the modality gap in multi-modal contrastive representation learning. Advances in Neural Information Processing Systems. 35, 17612\u201317625. https:\/\/doi.org\/10.48550\/arXiv.2203.02053","DOI":"10.48550\/arXiv.2203.02053"},{"key":"1019_CR15","doi-asserted-by":"publisher","unstructured":"Lin, D., Peng, Y.-X., & Meng, J., et al. (2024). Cross-modal adaptive dual association for text-to-image person retrieval. IEEE Transactions on Multimedia.,\u00a026, 6609\u20136620. https:\/\/doi.org\/10.1109\/TMM.2024.3355644","DOI":"10.1109\/TMM.2024.3355644"},{"key":"1019_CR16","doi-asserted-by":"publisher","unstructured":"Li, F., Wang, B., & Zhu, L., et al. (2024). Cross-domain transfer hashing for efficient cross-modal retrieval. IEEE Transactions on Circuits and Systems for Video Technology.,\u00a034(10), 9664\u20139677. https:\/\/doi.org\/10.1109\/TCSVT.2024.3374791","DOI":"10.1109\/TCSVT.2024.3374791"},{"key":"1019_CR17","doi-asserted-by":"publisher","unstructured":"Li, Z., Zhang, L., & Zhang, K., et al. (2024). Fast, accurate, and lightweight memory-enhanced embedding learning framework for image-text retrieval. IEEE Transactions on Circuits and Systems for Video Technology.,\u00a034(7), 6542\u20136558. https:\/\/doi.org\/10.1109\/TCSVT.2024.3358411","DOI":"10.1109\/TCSVT.2024.3358411"},{"key":"1019_CR18","doi-asserted-by":"publisher","unstructured":"Peng, Y., & Qi, J. (2019). Cm-gans: Cross-modal generative adversarial networks for common representation learning. ACM Transactions on Multimedia Computing, Communications, and Applications (TOMM),\u00a015(1), 1\u201324. https:\/\/doi.org\/10.1145\/3284750","DOI":"10.1145\/3284750"},{"key":"1019_CR19","doi-asserted-by":"publisher","unstructured":"Peng, Y., Qi, J., & Yuan, Y. (2018). Modality-specific cross-modal similarity measurement with recurrent attention network. IEEE Transactions on Image Processing.,\u00a027(11), 5585\u20135599. https:\/\/doi.org\/10.1109\/TIP.2018.2852503","DOI":"10.1109\/TIP.2018.2852503"},{"key":"1019_CR20","doi-asserted-by":"publisher","unstructured":"Qi, J., & Peng, Y. (2018). Cross-modal bidirectional translation via reinforcement learning. In: Proceedings of the Twenty-Seventh International Joint Conference on Artificial Intelligence, pp. 2630\u20132636. https:\/\/doi.org\/10.24963\/ijcai.2018\/365","DOI":"10.24963\/ijcai.2018\/365"},{"key":"1019_CR21","doi-asserted-by":"publisher","unstructured":"Radford, A., Kim, J. W., & Hallacy, C., et al. (2021). Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. https:\/\/doi.org\/10.48550\/arXiv.2103.00020","DOI":"10.48550\/arXiv.2103.00020"},{"key":"1019_CR22","doi-asserted-by":"publisher","unstructured":"Ranjan, V., Rasiwasia, N., & Jawahar, C. (2015). Multi-label cross-modal retrieval. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 4094\u20134102. https:\/\/doi.org\/10.1109\/ICCV.2015.466","DOI":"10.1109\/ICCV.2015.466"},{"key":"1019_CR23","unstructured":"Rashtchian, C., Young, P., & Hodosh, M., et al. (2010). Collecting image annotations using amazon\u2019s mechanical turk. In: Proceedings of the NAACL HLT 2010 Workshop on Creating Speech and Language Data with Amazon\u2019s Mechanical Turk, pp. 139\u2013147. https:\/\/aclanthology.org\/W10-0721\/"},{"key":"1019_CR24","doi-asserted-by":"publisher","unstructured":"Rasiwasia, N., Costa\u00a0Pereira, J., & Coviello, E., et al. (2010). A new approach to cross-modal multimedia retrieval. In: Proceedings of the 18th ACM International Conference on Multimedia, pp. 251\u2013260. https:\/\/doi.org\/10.1145\/1873951.1873987","DOI":"10.1145\/1873951.1873987"},{"key":"1019_CR25","doi-asserted-by":"publisher","unstructured":"Shi, D., Zhu, L., & Li, J., et al. (2024). Incomplete cross-modal retrieval with deep correlation transfer. ACM Transactions on Multimedia Computing, Communications and Applications.,\u00a020(5), 1\u201321. https:\/\/doi.org\/10.1145\/3637442","DOI":"10.1145\/3637442"},{"key":"1019_CR26","doi-asserted-by":"publisher","unstructured":"Sogi, N., Shibata, T., & Terao, M. (2024). Object-aware query perturbation for cross-modal image-text retrieval. In: European Conference on Computer Vision, pp. 447\u2013464. Springer, Cham. https:\/\/doi.org\/10.1007\/978-3-031-72986-7_26","DOI":"10.1007\/978-3-031-72986-7_26"},{"key":"1019_CR27","doi-asserted-by":"publisher","unstructured":"Tu, J., Liu, X., & Hao, Y., et al. (2024). Two-step discrete hashing for cross-modal retrieval. IEEE Transactions on Multimedia.,\u00a026, 8730\u20138741. https:\/\/doi.org\/10.1109\/TMM.2024.3381828","DOI":"10.1109\/TMM.2024.3381828"},{"key":"1019_CR28","doi-asserted-by":"publisher","unstructured":"Verma, Y., & Jawahar, C. (2014). Im2text and text2im: Associating images and texts for cross-modal retrieval. British Machine Vision Conference,\u00a01, 1\u201313. https:\/\/doi.org\/10.5244\/C.28.97","DOI":"10.5244\/C.28.97"},{"key":"1019_CR29","doi-asserted-by":"publisher","unstructured":"Wang, L., Li, Y., & Lazebnik, S. (2016). Learning deep structure-preserving image-text embeddings. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5005\u20135013. https:\/\/doi.org\/10.1109\/CVPR.2016.541","DOI":"10.1109\/CVPR.2016.541"},{"key":"1019_CR30","doi-asserted-by":"publisher","unstructured":"Wang, J., Song, Y., & Leung, T., et al. (2014). Learning fine-grained image similarity with deep ranking. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1386\u20131393. https:\/\/doi.org\/10.1109\/CVPR.2014.180","DOI":"10.1109\/CVPR.2014.180"},{"key":"1019_CR31","doi-asserted-by":"publisher","unstructured":"Wang, B., Yang, Y., & Xu, X., et al. (2017). Adversarial cross-modal retrieval. In: Proceedings of the 25th ACM International Conference on Multimedia, pp. 154\u2013162. https:\/\/doi.org\/10.1145\/3123266.3123326","DOI":"10.1145\/3123266.3123326"},{"key":"1019_CR32","doi-asserted-by":"publisher","unstructured":"Wang, Z., Wang, L., & Shi, Z., et al. (2024). A survey on person and vehicle re-identification. IET Computer Vision.,\u00a018(8), 1235\u20131268. https:\/\/doi.org\/10.1049\/cvi2.12316","DOI":"10.1049\/cvi2.12316"},{"key":"1019_CR33","doi-asserted-by":"publisher","unstructured":"Wei, Y., Zhao, Y., & Lu, C., et al. (2016). Cross-modal retrieval with cnn visual features: A new baseline. IEEE Transactions on Cybernetics.,\u00a047(2), 449\u2013460. https:\/\/doi.org\/10.1109\/TCYB.2016.2519449","DOI":"10.1109\/TCYB.2016.2519449"},{"key":"1019_CR34","doi-asserted-by":"publisher","unstructured":"Williams-Lekuona, M., & Cosma, G. (2025). Fico-itr: bridging fine-grained and coarse-grained image-text retrieval for comparative performance analysis. International Journal of Multimedia Information Retrieval.,\u00a014(2), 20. https:\/\/doi.org\/10.1007\/s13735-025-00368-6","DOI":"10.1007\/s13735-025-00368-6"},{"key":"1019_CR35","doi-asserted-by":"publisher","unstructured":"Xu, X., Lin, K., & Yang, Y., et al. (2020). Joint feature synthesis and embedding: Adversarial cross-modal retrieval revisited. IEEE Transactions on Pattern Analysis and Machine Intelligence.,\u00a044(6), 3030\u20133047. https:\/\/doi.org\/10.1109\/TPAMI.2020.3045530","DOI":"10.1109\/TPAMI.2020.3045530"},{"key":"1019_CR36","doi-asserted-by":"publisher","unstructured":"Zeng, Z., & Mao, W. (2022) A comprehensive empirical study of vision-language pre-trained model for supervised cross-modal retrieval. arXiv preprint arXiv:2201.02772. https:\/\/doi.org\/10.48550\/arXiv.2201.02772","DOI":"10.48550\/arXiv.2201.02772"},{"key":"1019_CR37","doi-asserted-by":"publisher","unstructured":"Zeng, Z., Sun, Y., & Mao, W. (2021). Mccn: Multimodal coordinated clustering network for large-scale cross-modal retrieval. In: Proceedings of the 29th ACM International Conference on Multimedia, pp. 5427\u20135435. https:\/\/doi.org\/10.1145\/3474085.3475670","DOI":"10.1145\/3474085.3475670"},{"key":"1019_CR38","doi-asserted-by":"publisher","unstructured":"Zeng, Z., He, S., & Zhang, Y., et al. (2025). A multimodal embedding transfer approach for consistent and selective learning processes in cross-modal retrieval. Information Sciences.,\u00a0704, Article 121974. https:\/\/doi.org\/10.1016\/j.ins.2025.121974","DOI":"10.1016\/j.ins.2025.121974"},{"key":"1019_CR39","doi-asserted-by":"publisher","unstructured":"Zhang, R. (2019). Making convolutional networks shift-invariant again. In: International Conference on Machine Learning, pp. 7324\u20137334. https:\/\/doi.org\/10.48550\/arXiv.1904.11486","DOI":"10.48550\/arXiv.1904.11486"},{"key":"1019_CR40","doi-asserted-by":"publisher","unstructured":"Zhang, Y., Ji, Z., & Wang, D., et al. (2024). User: Unified semantic enhancement with momentum contrast for image-text retrieval. IEEE Transactions on Image Processing.,\u00a033, 595\u2013609. https:\/\/doi.org\/10.1109\/TIP.2023.3348297","DOI":"10.1109\/TIP.2023.3348297"},{"issue":"5","key":"1019_CR41","doi-asserted-by":"publisher","first-page":"626","DOI":"10.1049\/cvi2.12268","volume":"18","author":"T Zhao","year":"2024","unstructured":"Zhao, T., Liu, P., & Lee, K. (2024). Omdet: Large-scale vision-language multi-dataset pre-training with multimodal detection network. IET Computer Vision.,18(5), 626\u2013639. https:\/\/doi.org\/10.1049\/cvi2.12268","journal-title":"IET Computer Vision."},{"key":"1019_CR42","doi-asserted-by":"publisher","unstructured":"Zhen, L., Hu, P., & Wang, X., et al. (2019). Deep supervised cross-modal retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10394\u201310403. https:\/\/doi.org\/10.1109\/CVPR.2019.01064","DOI":"10.1109\/CVPR.2019.01064"}],"container-title":["Journal of Intelligent Information Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10844-025-01019-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10844-025-01019-2","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10844-025-01019-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T02:45:38Z","timestamp":1778035538000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10844-025-01019-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1,7]]},"references-count":42,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026,4]]}},"alternative-id":["1019"],"URL":"https:\/\/doi.org\/10.1007\/s10844-025-01019-2","relation":{},"ISSN":["0925-9902","1573-7675"],"issn-type":[{"value":"0925-9902","type":"print"},{"value":"1573-7675","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,1,7]]},"assertion":[{"value":"19 September 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 December 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 December 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 January 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of Interest\/Competing Interests"}},{"value":"The authors declare no competing interests.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}]}}