{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T04:31:04Z","timestamp":1781584264326,"version":"3.54.5"},"reference-count":43,"publisher":"Springer Science and Business Media LLC","issue":"40","license":[{"start":{"date-parts":[[2024,1,12]],"date-time":"2024-01-12T00:00:00Z","timestamp":1705017600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,12]],"date-time":"2024-01-12T00:00:00Z","timestamp":1705017600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-023-17956-5","type":"journal-article","created":{"date-parts":[[2024,1,12]],"date-time":"2024-01-12T06:01:51Z","timestamp":1705039311000},"page":"88221-88243","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Semantic enhancement and multi-level alignment network for cross-modal retrieval"],"prefix":"10.1007","volume":"83","author":[{"given":"Jia","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hong","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,1,12]]},"reference":[{"issue":"12","key":"17956_CR1","doi-asserted-by":"publisher","first-page":"1551","DOI":"10.1631\/FITEE.2100463","volume":"22","author":"Y Yang","year":"2021","unstructured":"Yang Y, Zhuang Y, Pan Y (2021) Multiple knowledge representation for big data artificial intelligence: framework, applications, and case studies. Front Inform Technol Electron Eng 22(12):1551\u20131558","journal-title":"Front Inform Technol Electron Eng"},{"issue":"3","key":"17956_CR2","first-page":"489","volume":"16","author":"L Ying","year":"2022","unstructured":"Ying L, Yingying G, Jie F, Jiulun F, Yu H, Jiming L (2022) Survey of research on deep learning image-text cross-modal retrieval. J Front Comput Sci Technol 16(3):489","journal-title":"J Front Comput Sci Technol"},{"issue":"4","key":"17956_CR3","doi-asserted-by":"publisher","first-page":"041401","DOI":"10.1115\/1.4056436","volume":"145","author":"X Li","year":"2023","unstructured":"Li X, Wang Y, Sha Z (2023) Deep learning methods of cross-modal tasks for conceptual design of product shapes: A review. J Mech Des 145(4):041401","journal-title":"J Mech Des"},{"key":"17956_CR4","doi-asserted-by":"publisher","first-page":"6079","DOI":"10.1109\/TMM.2022.3204444","volume":"25","author":"X Wang","year":"2022","unstructured":"Wang X, Zhu L, Zheng Z, Xu M, Yang Y (2022) Align and tell: Boosting text-video retrieval with local alignment and fine-grained supervision. IEEE Trans Multimedia 25:6079\u20136089. https:\/\/doi.org\/10.1109\/TMM.2022.3204444","journal-title":"IEEE Trans Multimedia"},{"issue":"6","key":"17956_CR5","doi-asserted-by":"publisher","first-page":"6605","DOI":"10.1109\/TPAMI.2020.3015894","volume":"45","author":"X Wang","year":"2020","unstructured":"Wang X, Zhu L, Wu Y, Yang Y (2020) Symbiotic attention for egocentric action recognition with object-centric alignment. IEEE Transactions on Pattern Analysis and Machine Intelligence 45(6):6605\u20136617. https:\/\/doi.org\/10.1109\/TPAMI.2020.3015894","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"17956_CR6","doi-asserted-by":"crossref","unstructured":"Cao M, Li S, Li J, Nie L, Zhang M (2022) Image-text retrieval: A survey on recent research and development. arXiv preprint arXiv:2203.14713","DOI":"10.24963\/ijcai.2022\/759"},{"issue":"5","key":"17956_CR7","doi-asserted-by":"publisher","first-page":"2465","DOI":"10.1109\/TCSVT.2022.3220297","volume":"33","author":"Z Liu","year":"2022","unstructured":"Liu Z, Chen F, Xu J, Pei W, Lu G (2022) Image-Text Retrieval with Cross-Modal Semantic Importance Consistency. IEEE Transactions on Circuits and Systems for Video Technology 33(5):2465\u20132476. https:\/\/doi.org\/10.1109\/TCSVT.2022.3220297","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"17956_CR8","doi-asserted-by":"crossref","unstructured":"Zhang K, Mao Z, Wang Q, Zhang Y (2022) Negative-aware attention framework for image-text matching. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 15661\u201315670","DOI":"10.1109\/CVPR52688.2022.01521"},{"issue":"1","key":"17956_CR9","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3284750","volume":"15","author":"Y Peng","year":"2019","unstructured":"Peng Y, Qi J (2019) CM-GANs: Cross-modal generative adversarial networks for common representation learning. ACM Trans Multimed Comput Commun Appl 15(1):1\u201324","journal-title":"ACM Trans Multimed Comput Commun Appl"},{"key":"17956_CR10","doi-asserted-by":"publisher","first-page":"9189","DOI":"10.1109\/TMM.2023.3248160","volume":"25","author":"J Guo","year":"2023","unstructured":"Guo J et al (2023) (2023) HGAN: Hierarchical Graph Alignment Network for Image-Text Retrieval. IEEE Trans Multimedia 25:9189\u20139202. https:\/\/doi.org\/10.1109\/TMM.2023.3248160","journal-title":"IEEE Trans Multimedia"},{"key":"17956_CR11","unstructured":"Frome A et al (2013) Devise: A deep visual-semantic embedding model. Adv Neural Inform Process Syst 26"},{"key":"17956_CR12","unstructured":"Faghri F, Fleet DJ, Kiros JR, Fidler S (2017) Vse++: Improving visual-semantic embeddings with hard negatives. arXiv preprint arXiv:1707.05612"},{"key":"17956_CR13","doi-asserted-by":"crossref","unstructured":"Gu J, Cai J, Joty SR, Niu L, Wang G (2018) Look, imagine and match: Improving textual-visual cross-modal retrieval with generative models. In Proceedings of the IEEE conference on computer vision and pattern recognition, pp 7181\u20137189","DOI":"10.1109\/CVPR.2018.00750"},{"key":"17956_CR14","doi-asserted-by":"crossref","unstructured":"Zhen L, Hu P, Wang X, Peng D (2019) Deep supervised cross-modal retrieval. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 10394\u201310403","DOI":"10.1109\/CVPR.2019.01064"},{"issue":"7","key":"17956_CR15","doi-asserted-by":"publisher","first-page":"2866","DOI":"10.1109\/TCSVT.2020.3030656","volume":"31","author":"K Wen","year":"2020","unstructured":"Wen K, Gu X, Cheng Q (2020) Learning dual semantic relations with graph attention for image-text matching. IEEE Trans Circ Syst Video Technol 31(7):2866\u20132879","journal-title":"IEEE Trans Circ Syst Video Technol"},{"key":"17956_CR16","doi-asserted-by":"crossref","unstructured":"Chen J, Hu H, Wu H, Jiang Y, Wang C (2021) Learning the best pooling strategy for visual semantic embedding. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 15789\u201315798","DOI":"10.1109\/CVPR46437.2021.01553"},{"issue":"4","key":"17956_CR17","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3572844","volume":"19","author":"S Yang","year":"2023","unstructured":"Yang S et al (2023) Semantic Completion and Filtration for Image-Text Retrieval. ACM Trans Multimed Comput Commun Appl 19(4):1\u201320","journal-title":"ACM Trans Multimed Comput Commun Appl"},{"key":"17956_CR18","doi-asserted-by":"publisher","first-page":"9193","DOI":"10.1109\/TIP.2021.3123553","volume":"30","author":"J Li","year":"2021","unstructured":"Li J, Liu L, Niu L, Zhang L (2021) Memorize, associate and match: Embedding enhancement via fine-grained alignment for image-text retrieval. IEEE Trans Image Process 30:9193\u20139207","journal-title":"IEEE Trans Image Process"},{"key":"17956_CR19","doi-asserted-by":"crossref","unstructured":"Ling Z, Xing Z, Li J, Niu L (2022) Multi-level region matching for fine-grained sketch-based image retrieval. In Proceedings of the 30th ACM International Conference on Multimedia, pp 462\u2013470","DOI":"10.1145\/3503161.3548147"},{"key":"17956_CR20","unstructured":"Karpathy A, Joulin A, Fei-Fei LF (2014) Deep fragment embeddings for bidirectional image sentence mapping.\u00a0Advances Neural Inform Process Syst\u00a027"},{"key":"17956_CR21","doi-asserted-by":"crossref","unstructured":"Niu Z, Zhou M, Wang L, Gao X, Hua G (2017) Hierarchical multimodal lstm for dense visual-semantic embedding. In: Proceedings of the IEEE International Conference on Computer Vision, pp 1881\u20131889","DOI":"10.1109\/ICCV.2017.208"},{"key":"17956_CR22","doi-asserted-by":"crossref","unstructured":"Nam H, Ha J-W, Kim J (2017) Dual attention networks for multimodal reasoning and matching. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 299\u2013307","DOI":"10.1109\/CVPR.2017.232"},{"key":"17956_CR23","doi-asserted-by":"crossref","unstructured":"Lee K-H, Chen X, Hua G, Hu H, He X (2018) Stacked cross attention for image-text matching. In Proceedings of the European Conference on Computer Vision (ECCV), pp 201\u2013216","DOI":"10.1007\/978-3-030-01225-0_13"},{"key":"17956_CR24","doi-asserted-by":"crossref","unstructured":"Chen H, Ding G, Liu X, Lin Z, Liu J, Han J (2020) Imram: Iterative matching with recurrent attention memory for cross-modal image-text retrieval. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern recOgnition, pp 12655\u201312663","DOI":"10.1109\/CVPR42600.2020.01267"},{"key":"17956_CR25","doi-asserted-by":"crossref","unstructured":". Qu L, Liu M, Wu J, Gao Z, Nie L (2021) Dynamic modality interaction modeling for image-text retrieval. In Proceedings of the 44th International ACM SIGIR Conference on Research and Development in Information Retrieval, pp 1104\u20131113","DOI":"10.1145\/3404835.3462829"},{"key":"17956_CR26","doi-asserted-by":"crossref","unstructured":"Ji Z, Chen K, Wang H (2021) Step-wise hierarchical alignment network for image-text matching. arXiv preprint arXiv:2106.06509","DOI":"10.24963\/ijcai.2021\/106"},{"issue":"11","key":"17956_CR27","doi-asserted-by":"publisher","first-page":"8037","DOI":"10.1109\/TCSVT.2022.3182426","volume":"32","author":"S Yang","year":"2022","unstructured":"Yang S, Li Q, Li W, Li X, Liu A-A (2022) Dual-Level Representation Enhancement on Characteristic and Context for Image-Text Retrieval. IEEE Trans Circ Syst Video Technol 32(11):8037\u20138050","journal-title":"IEEE Trans Circ Syst Video Technol"},{"key":"17956_CR28","unstructured":"Xiao Y et al (2023) Local-Global Temporal Difference Learning for Satellite Video Super-Resolution. arXiv preprint arXiv:2304.04421"},{"key":"17956_CR29","doi-asserted-by":"crossref","unstructured":"Jiang K, Wang Z, Chen C, Wang Z, Cui L, Lin C-W (2022) Magic ELF: Image deraining meets association learning and transformer. arXiv preprint arXiv:2207.10455","DOI":"10.1145\/3503161.3547760"},{"key":"17956_CR30","doi-asserted-by":"crossref","unstructured":"Anderson P et al (2018) Bottom-up and top-down attention for image captioning and visual question answering. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 6077\u20136086","DOI":"10.1109\/CVPR.2018.00636"},{"key":"17956_CR31","unstructured":"Ren S, He K, Girshick R, Sun J (2015) Faster r-cnn: Towards real-time object detection with region proposal networks.\u00a0Adv Neural Inform Process Syst 28"},{"key":"17956_CR32","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"17956_CR33","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","volume":"123","author":"R Krishna","year":"2017","unstructured":"Krishna R et al (2017) Visual genome: Connecting language and vision using crowdsourced dense image annotations. Int J Comput Vision 123:32\u201373","journal-title":"Int J Comput Vision"},{"key":"17956_CR34","doi-asserted-by":"crossref","unstructured":"Deng J, Dong W, Socher R, Li L-J, Li K, Fei-Fei L (2009) Imagenet: A large-scale hierarchical image database. In 2009 IEEE Conference on Computer Vision and Pattern Recognition 9, pp. 248\u2013255: Ieee","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"17956_CR35","unstructured":"Bahdanau D, Cho K, Bengio Y (2014) Neural machine translation by jointly learning to align and translate. arXiv preprint arXiv:1409.0473"},{"issue":"11","key":"17956_CR36","doi-asserted-by":"publisher","first-page":"2673","DOI":"10.1109\/78.650093","volume":"45","author":"M Schuster","year":"1997","unstructured":"Schuster M, Paliwal KK (1997) Bidirectional recurrent neural networks. IEEE Trans Signal Process 45(11):2673\u20132681","journal-title":"IEEE Trans Signal Process"},{"key":"17956_CR37","unstructured":"Vaswani A et al (2017) Attention is all you need.\u00a0Adv Neural Inform Process Syst 30"},{"key":"17956_CR38","unstructured":"Glorot X, Bengio Y (2010) Understanding the difficulty of training deep feedforward neural networks. in Proceedings of the thirteenth international conference on artificial intelligence and statistics, pp 249\u2013256: JMLR Workshop and Conference Proceedings"},{"key":"17956_CR39","unstructured":"Kingma DP, Ba J (2014) Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980"},{"key":"17956_CR40","doi-asserted-by":"crossref","unstructured":"Plummer BA, Wang L, Cervantes CM, Caicedo JC, Hockenmaier J Lazebnik S (2015) Flickr30k entities: Collecting region-to-phrase correspondences for richer image-to-sentence models. In Proceedings of the IEEE international conference on computer vision, pp 2641\u20132649","DOI":"10.1109\/ICCV.2015.303"},{"issue":"4","key":"17956_CR41","doi-asserted-by":"publisher","first-page":"652","DOI":"10.1109\/TPAMI.2016.2587640","volume":"39","author":"O Vinyals","year":"2016","unstructured":"Vinyals O, Toshev A, Bengio S, Erhan D (2016) Show and tell: Lessons learned from the 2015 mscoco image captioning challenge. IEEE Trans Pattern Anal Mach Intell 39(4):652\u2013663","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"issue":"5","key":"17956_CR42","first-page":"4527","volume":"35","author":"S He","year":"2022","unstructured":"He S et al (2022) Category alignment adversarial learning for cross-modal retrieval. IEEE Trans Knowl Data Eng 35(5):4527\u20134538","journal-title":"IEEE Trans Knowl Data Eng"},{"key":"17956_CR43","doi-asserted-by":"publisher","first-page":"103807","DOI":"10.1016\/j.jvcir.2023.103807","volume":"93","author":"M Yuan","year":"2023","unstructured":"Yuan M, Zhang H, Liu D, Wang L, Liu L (2023) Semantic-embedding Guided Graph Network for cross-modal retrieval. J. Vis. Commun. Image Represent. 93:103807","journal-title":"J. Vis. Commun. Image Represent."}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-17956-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-023-17956-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-17956-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,13]],"date-time":"2024-12-13T10:11:59Z","timestamp":1734084719000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-023-17956-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,1,12]]},"references-count":43,"journal-issue":{"issue":"40","published-online":{"date-parts":[[2024,12]]}},"alternative-id":["17956"],"URL":"https:\/\/doi.org\/10.1007\/s11042-023-17956-5","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,1,12]]},"assertion":[{"value":"9 June 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 December 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 December 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 January 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared toinfluence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}]}}