{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T04:26:52Z","timestamp":1772771212614,"version":"3.50.1"},"reference-count":33,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2025,1,6]],"date-time":"2025-01-06T00:00:00Z","timestamp":1736121600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,6]],"date-time":"2025-01-06T00:00:00Z","timestamp":1736121600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"DOI":"10.1007\/s11227-024-06840-0","type":"journal-article","created":{"date-parts":[[2025,1,6]],"date-time":"2025-01-06T06:28:57Z","timestamp":1736144937000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Syntactic-guided optimization of image\u2013text matching for intra-modal modeling"],"prefix":"10.1007","volume":"81","author":[{"given":"Di","family":"Wu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Le","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yao","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,1,6]]},"reference":[{"issue":"8","key":"6840_CR1","doi-asserted-by":"publisher","first-page":"8611","DOI":"10.1007\/s11227-022-05001-5","volume":"79","author":"Y Yi","year":"2023","unstructured":"Yi Y, Tian Y, He C, Fan Y, Hu X, Xu Y (2023) Dbt: multimodal emotion recognition based on dual-branch transformer. J Supercomput 79(8):8611\u20138633","journal-title":"J Supercomput"},{"issue":"16","key":"6840_CR2","doi-asserted-by":"publisher","first-page":"17810","DOI":"10.1007\/s11227-023-05318-9","volume":"79","author":"X Shi","year":"2023","unstructured":"Shi X, Yu Z, Wang X, Li Y, Niu Y (2023) Text-image matching for multi-model machine translation. J Supercomput 79(16):17810\u201317823","journal-title":"J Supercomput"},{"issue":"7","key":"6840_CR3","doi-asserted-by":"publisher","first-page":"7916","DOI":"10.1007\/s11227-022-04912-7","volume":"79","author":"M Kayani","year":"2023","unstructured":"Kayani M, Ghafoor A, Riaz MM (2023) Multi-modal text recognition and encryption in scanned document images. J Supercomput 79(7):7916\u20137936","journal-title":"J Supercomput"},{"key":"6840_CR4","first-page":"1218","volume":"35","author":"H Diao","year":"2021","unstructured":"Diao H, Zhang Y, Ma L, Lu H (2021) Similarity reasoning and filtration for image-text matching. Proc AAAI Conf Artif Intell 35:1218\u20131226","journal-title":"Proc AAAI Conf Artif Intell"},{"issue":"12","key":"6840_CR5","doi-asserted-by":"publisher","first-page":"2639","DOI":"10.1162\/0899766042321814","volume":"16","author":"DR Hardoon","year":"2004","unstructured":"Hardoon DR, Szedmak S, Shawe-Taylor J (2004) Canonical correlation analysis: an overview with application to learning methods. Neural Comput 16(12):2639\u20132664","journal-title":"Neural Comput"},{"key":"6840_CR6","unstructured":"Kiros R, Salakhutdinov R, Zemel RS (2014) Unifying visual-semantic embeddings with multimodal neural language models. arXiv preprint arXiv:1411.2539"},{"key":"6840_CR7","doi-asserted-by":"crossref","unstructured":"Ma L, Lu Z, Shang L, Li H (2015) Multimodal convolutional neural networks for matching image and sentence. In: Proceedings of the IEEE International Conference on Computer Vision, pp 2623\u20132631","DOI":"10.1109\/ICCV.2015.301"},{"key":"6840_CR8","first-page":"200","volume":"48","author":"L Liu","year":"2021","unstructured":"Liu L, Gou T (2021) Cross-modal retrieval combining deep canonical correlation analysis and adversarial learning. Comput Sci 48:200\u2013207","journal-title":"Comput Sci"},{"key":"6840_CR9","doi-asserted-by":"crossref","unstructured":"Ge X, Chen F, Xu S, Tao F, Jose JM (2023) Cross-modal semantic enhanced interaction for image-sentence retrieval. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp 1022\u20131031","DOI":"10.1109\/WACV56688.2023.00108"},{"key":"6840_CR10","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2023.110505","volume":"270","author":"Z Wang","year":"2023","unstructured":"Wang Z, Xu X, Wei J, Xie N, Shao J, Yang Y (2023) Quaternion representation learning for cross-modal matching. Knowl-Based Syst 270:110505","journal-title":"Knowl-Based Syst"},{"key":"6840_CR11","doi-asserted-by":"crossref","unstructured":"Xu G, Hu M, Wang X Yang J, Li N, Zhang Q (2023) Location attention knowledge embedding model for image-text matching. In: Chinese Conference on Pattern Recognition and Computer Vision (PRCV), pp 408\u2013421 Springer","DOI":"10.1007\/978-981-99-8429-9_33"},{"issue":"4","key":"6840_CR12","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3499027","volume":"18","author":"Y Cheng","year":"2022","unstructured":"Cheng Y, Zhu X, Qian J, Wen F, Liu P (2022) Cross-modal graph matching network for image-text retrieval. ACM Trans Multimed Comput, Commun, Appl (TOMM) 18(4):1\u201323","journal-title":"ACM Trans Multimed Comput, Commun, Appl (TOMM)"},{"issue":"1","key":"6840_CR13","doi-asserted-by":"publisher","first-page":"641","DOI":"10.1109\/TPAMI.2022.3148470","volume":"45","author":"K Li","year":"2022","unstructured":"Li K, Zhang Y, Li K, Li Y, Fu Y (2022) Image-text embedding learning via visual and textual semantic reasoning. IEEE Trans Pattern Anal Mach Intell 45(1):641\u2013656","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"6840_CR14","unstructured":"Li Y, Yao T, Zhang L, Sun Y, Fu H (2024) Image-text matching algorithm based on multi-level semantic alignment. J Beijing Univ Aeronaut Astronaut 50:551\u2013558. https:\/\/doi.org\/10.13700\/j.bh.1001-5965.2022.0385"},{"key":"6840_CR15","first-page":"3262","volume":"36","author":"H Zhang","year":"2022","unstructured":"Zhang H, Mao Z, Zhang K, Zhang Y (2022) Show your faith: cross-modal confidence-aware network for image-text matching. Proc AAAI Conf Artif Intell 36:3262\u20133270","journal-title":"Proc AAAI Conf Artif Intell"},{"issue":"2","key":"6840_CR16","doi-asserted-by":"publisher","first-page":"4383","DOI":"10.1007\/s11042-023-15321-0","volume":"83","author":"X Qin","year":"2024","unstructured":"Qin X, Li L, Pang G (2024) Multi-scale motivated neural network for image-text matching. Multimed Tools Appl 83(2):4383\u20134407","journal-title":"Multimed Tools Appl"},{"key":"6840_CR17","doi-asserted-by":"publisher","first-page":"2973","DOI":"10.1109\/TCSVT.2023.3307554","volume":"34","author":"K Zhang","year":"2023","unstructured":"Zhang K, Hu B, Zhang H, Li Z, Mao Z (2023) Enhanced semantic similarity learning framework for image-text matching. IEEE Trans Circuits Syst Video Technol 34:2973\u20132988","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"issue":"4","key":"6840_CR18","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3631356","volume":"20","author":"T Yao","year":"2023","unstructured":"Yao T, Li Y, Li Y, Zhu Y, Wang G, Yue J (2023) Cross-modal semantically augmented network for image-text matching. ACM Trans Multimed Comput Commun Appl 20(4):1\u201318","journal-title":"ACM Trans Multimed Comput Commun Appl"},{"key":"6840_CR19","doi-asserted-by":"crossref","unstructured":"Jiang D, Ye M (2023) Cross-modal implicit relation reasoning and aligning for text-to-image person retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 2787\u20132797","DOI":"10.1109\/CVPR52729.2023.00273"},{"key":"6840_CR20","doi-asserted-by":"crossref","unstructured":"Chen C, Zhang B, Cao L, Shen J, Gunter T, Jose AM, Toshev A, Shlens J, Pang R, Yang Y (2023) Stair: Learning sparse text and image representation in grounded tokens. arXiv preprint arXiv:2301.13081","DOI":"10.18653\/v1\/2023.emnlp-main.932"},{"issue":"2","key":"6840_CR21","doi-asserted-by":"publisher","first-page":"948","DOI":"10.1109\/TCYB.2022.3179020","volume":"54","author":"X Liu","year":"2022","unstructured":"Liu X, He Y, Cheung Y-M, Xu X, Wang N (2022) Learning relationship-enhanced semantic graph for fine-grained image-text matching. IEEE Trans Cybern 54(2):948\u2013961","journal-title":"IEEE Trans Cybern"},{"issue":"3","key":"6840_CR22","doi-asserted-by":"publisher","first-page":"41","DOI":"10.1109\/MIS.2023.3265176","volume":"38","author":"H Shang","year":"2023","unstructured":"Shang H, Zhao G, Shi J, Qian X (2023) Multi-view text imagination network based on latent alignment for image-text matching. IEEE Intell Syst 38(3):41\u201350","journal-title":"IEEE Intell Syst"},{"key":"6840_CR23","doi-asserted-by":"crossref","unstructured":"Fu Z, Mao Z, Song Y, Zhang Y (2023) Learning semantic relationship among instances for image-text matching. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 15159\u201315168","DOI":"10.1109\/CVPR52729.2023.01455"},{"key":"6840_CR24","doi-asserted-by":"crossref","unstructured":"Long S, Han SC, Wan X, Poon J (2022) Gradual: Graph-based dual-modal representation for image-text matching. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp 3459\u20133468","DOI":"10.1109\/WACV51458.2022.00252"},{"issue":"5","key":"6840_CR25","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3563390","volume":"22","author":"J Pei","year":"2023","unstructured":"Pei J, Zhong K, Yu Z, Wang L, Lakshmanna K (2023) Scene graph semantic inference for image and text matching. ACM Trans Asian Low-Resour Lang Inf Process 22(5):1\u201323","journal-title":"ACM Trans Asian Low-Resour Lang Inf Process"},{"issue":"9","key":"6840_CR26","doi-asserted-by":"publisher","first-page":"6437","DOI":"10.1109\/TCSVT.2022.3164230","volume":"32","author":"X Dong","year":"2022","unstructured":"Dong X, Zhang H, Zhu L, Nie L, Liu L (2022) Hierarchical feature aggregation based on transformer for image-text matching. IEEE Trans Circuits Syst Video Technol 32(9):6437\u20136447","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"key":"6840_CR27","doi-asserted-by":"crossref","unstructured":"Lee K-H, Chen X, Hua G, Hu H, He X (2018) Stacked cross attention for image-text matching. In: Proceedings of the European Conference on Computer Vision (ECCV), pp 201\u2013216","DOI":"10.1007\/978-3-030-01225-0_13"},{"issue":"6","key":"6840_CR28","first-page":"66","volume":"10","author":"Y Salini","year":"2023","unstructured":"Salini Y, Eswaraiah P, Brahmam MV, Sirisha U (2023) Word embedding for text classification: Efficient cnn and bi-gru fusion multi attention mechanism. EAI Endorsed Trans Scalable Inf Syst 10(6):66","journal-title":"EAI Endorsed Trans Scalable Inf Syst"},{"key":"6840_CR29","doi-asserted-by":"crossref","unstructured":"Zhou J, Zhao H (2019) Head-driven phrase structure grammar parsing on penn treebank. In: Proceedings of the 57th Conference of the Association for Computational Linguistics, ACL 2019, Florence, Italy, July 28- August 2, 2019, Volume 1: Long Papers, pp 2396\u20132408","DOI":"10.18653\/v1\/P19-1230"},{"key":"6840_CR30","first-page":"1143","volume":"16","author":"D Gong","year":"2021","unstructured":"Gong D, Chen H, Chen S, Bao Y, Ding G (2021) Matching with agreement for cross-modal image-text retrieval. CAAI Trans Intell Syst 16:1143\u20131150","journal-title":"CAAI Trans Intell Syst"},{"issue":"3","key":"6840_CR31","doi-asserted-by":"publisher","first-page":"1057","DOI":"10.1007\/s00530-022-01038-x","volume":"29","author":"H Sun","year":"2023","unstructured":"Sun H, Qin X, Liu X (2023) Image-text matching using multi-subspace joint representation. Multimed Syst 29(3):1057\u20131071","journal-title":"Multimed Syst"},{"key":"6840_CR32","doi-asserted-by":"crossref","unstructured":"Wei X, Zhang T, Li Y, Zhang Y, Wu F (2020) Multi-modality cross attention network for image and sentence matching. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 10941\u201310950","DOI":"10.1109\/CVPR42600.2020.01095"},{"issue":"2","key":"6840_CR33","doi-asserted-by":"publisher","first-page":"948","DOI":"10.1109\/TCYB.2022.3179020","volume":"54","author":"X Liu","year":"2024","unstructured":"Liu X, He Y, Cheung Y, Xu X, Wang N (2024) Learning relationship-enhanced semantic graph for fine-grained image-text matching. IEEE Trans Cybern 54(2):948\u2013961","journal-title":"IEEE Trans Cybern"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-024-06840-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-024-06840-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-024-06840-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,6]],"date-time":"2025-01-06T07:13:17Z","timestamp":1736147597000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-024-06840-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,1,6]]},"references-count":33,"journal-issue":{"issue":"2","published-online":{"date-parts":[[2025,1]]}},"alternative-id":["6840"],"URL":"https:\/\/doi.org\/10.1007\/s11227-024-06840-0","relation":{},"ISSN":["1573-0484"],"issn-type":[{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,1,6]]},"assertion":[{"value":"17 December 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 January 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"This article does not contain any studies with human participants or animals performed by any of the authors.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval"}},{"value":"Informed consent was obtained from all individual participants included in the study.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}},{"value":"Publication consent was obtained from all individual participants included in the study.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}}],"article-number":"367"}}