{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,7,8]],"date-time":"2025-07-08T09:05:33Z","timestamp":1751965533687,"version":"3.37.3"},"reference-count":55,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2023,1,7]],"date-time":"2023-01-07T00:00:00Z","timestamp":1673049600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,1,7]],"date-time":"2023-01-07T00:00:00Z","timestamp":1673049600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61728204","61728204","61728204"],"award-info":[{"award-number":["61728204","61728204","61728204"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2023,6]]},"DOI":"10.1007\/s00530-022-01038-x","type":"journal-article","created":{"date-parts":[[2023,1,7]],"date-time":"2023-01-07T04:26:36Z","timestamp":1673065596000},"page":"1057-1071","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Image-text matching using multi-subspace joint representation"],"prefix":"10.1007","volume":"29","author":[{"given":"Hao","family":"Sun","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaolin","family":"Qin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaojing","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,1,7]]},"reference":[{"key":"1038_CR1","doi-asserted-by":"publisher","first-page":"321","DOI":"10.1093\/biomet\/28.3-4.321","volume":"28","author":"H Hotelling","year":"1935","unstructured":"Hotelling, H.: Relations between two sets of variates. Biometrika 28, 321\u2013377 (1935)","journal-title":"Biometrika"},{"key":"1038_CR2","unstructured":"Akaho, S.: A kernel method for canonical correlation analysis. arXiv preprint cs\/0609071 (2006)"},{"issue":"2","key":"1038_CR3","doi-asserted-by":"publisher","first-page":"257","DOI":"10.1162\/NECO_a_00801","volume":"28","author":"S Chandar","year":"2016","unstructured":"Chandar, S., Khapra, M.M., Larochelle, H., Ravindran, B.: Correlational neural networks. Neural Comput. 28(2), 257 (2016)","journal-title":"Neural Comput."},{"key":"1038_CR4","doi-asserted-by":"publisher","unstructured":"Yan, F., Mikolajczyk, K.: Deep correlation for matching images and text. 2015 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) 3441\u20133450 (2015). https:\/\/doi.org\/10.1109\/CVPR.2015.7298966","DOI":"10.1109\/CVPR.2015.7298966"},{"key":"1038_CR5","unstructured":"Andrew, G., Arora, R., Bilmes, J., Livescu, K.: Deep canonical correlation analysis. Proceedings of the 30th International Conference on International Conference on Machine Learning 28, III-1247-III-1255 (2013)"},{"key":"1038_CR6","doi-asserted-by":"crossref","unstructured":"Li, K., Zhang, Y., Li, K., Li, Y., Fu, Y.: Visual semantic reasoning for image-text matching. Proceedings of the IEEE\/CVF International Conference on Computer Vision 4654\u20134662 (2019)","DOI":"10.1109\/ICCV.2019.00475"},{"key":"1038_CR7","doi-asserted-by":"crossref","unstructured":"Li, J., Niu, L., Zhang, L.: Action-aware embedding enhancement for image-text retrieval. Thirty-Sixth AAAI Conference on Artificial Intelligence, AAAI 2022, Thirty-Fourth Conference on Innovative Applications of Artificial Intelligence, IAAI 2022, The Twelveth Symposium on Educational Advances in Artificial Intelligence, EAAI 2022 Virtual Event, February 22 - March 1, 2022 1323\u20131331 (2022). https:\/\/ojs.aaai.org\/index.php\/AAAI\/article\/view\/20020","DOI":"10.1609\/aaai.v36i2.20020"},{"key":"1038_CR8","doi-asserted-by":"publisher","unstructured":"Tan, W., Zhu, L., Guan, W., Li, J., Cheng, Z.: Bit-aware semantic transformer hashing for multi-modal retrieval. Proceedings of the 45th International ACM SIGIR Conference on Research and Development in Information Retrieval 982-991 (2022).https:\/\/doi.org\/10.1145\/3477495.3531947","DOI":"10.1145\/3477495.3531947"},{"key":"1038_CR9","unstructured":"Karpathy, A., Joulin, A., Fei-Fei, L.: Deep fragment embeddings for bidirectional image sentence mapping. arXiv preprint arXiv:1406.5679 (2014)"},{"key":"1038_CR10","doi-asserted-by":"crossref","unstructured":"Wu, Y., Wang, S., Song, G., Huang, Q.: Learning fragment self-attention embeddings for image-text matching. Proceedings of the 27th ACM International Conference on Multimedia 2088\u20132096 (2019)","DOI":"10.1145\/3343031.3350940"},{"key":"1038_CR11","unstructured":"Vendrov, I., Kiros, R., Fidler, S., Urtasun, R.: Order-embeddings of images and language. arXiv preprint arXiv:1511.06361 (2015)"},{"key":"1038_CR12","doi-asserted-by":"crossref","unstructured":"Semedo, D., Magalh\u00e3es, J.: Cross-modal subspace learning with scheduled adaptive margin constraints. Proceedings of the 27th ACM International Conference on Multimedia 75\u201383 (2019)","DOI":"10.1145\/3343031.3351030"},{"issue":"2","key":"1038_CR13","doi-asserted-by":"publisher","first-page":"394","DOI":"10.1109\/TPAMI.2018.2797921","volume":"41","author":"L Wang","year":"2018","unstructured":"Wang, L., Li, Y., Huang, J., Lazebnik, S.: Learning two-branch neural networks for image-text matching tasks. IEEE Trans. Pattern Anal. Mach. Intell. 41(2), 394\u2013407 (2018)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1038_CR14","doi-asserted-by":"crossref","unstructured":"Wang, L., Li, Y., Lazebnik, S.: Learning deep structure-preserving image-text embeddings. Proceedings of the IEEE conference on computer vision and pattern recognition 5005\u20135013 (2016)","DOI":"10.1109\/CVPR.2016.541"},{"key":"1038_CR15","doi-asserted-by":"publisher","unstructured":"Zhong, X.: et al. Auxiliary bi-level graph representation for cross-modal image-text retrieval. 2021 IEEE International Conference on Multimedia and Expo (ICME) 1\u20136 (2021). https:\/\/doi.org\/10.1109\/ICME51207.2021.9428380","DOI":"10.1109\/ICME51207.2021.9428380"},{"issue":"1","key":"1038_CR16","doi-asserted-by":"publisher","first-page":"388","DOI":"10.1109\/TCSVT.2021.3060713","volume":"32","author":"J Wu","year":"2022","unstructured":"Wu, J., Wu, C., Lu, J., Wang, L., Cui, X.: Region reinforcement network with topic constraint for image-text matching. IEEE Trans. Cir. and Sys. for Video Technol. 32(1), 388\u2013397 (2022). https:\/\/doi.org\/10.1109\/TCSVT.2021.3060713","journal-title":"IEEE Trans. Cir. and Sys. for Video Technol."},{"key":"1038_CR17","doi-asserted-by":"publisher","unstructured":"Ge, X.: et al. Structured multi-modal feature embedding and alignment for image-sentence retrieval. Proceedings of the 29th ACM International Conference on Multimedia 5185-5193 (2021).https:\/\/doi.org\/10.1145\/3474085.3475634","DOI":"10.1145\/3474085.3475634"},{"key":"1038_CR18","unstructured":"Faghri, F., Fleet, D.J., Kiros, J.R., Fidler, S.: Vse++: Improving visual-semantic embeddings with hard negatives. arXiv preprint arXiv:1707.05612 (2017)"},{"key":"1038_CR19","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Lu, H.: Deep cross-modal projection learning for image-text matching. Proceedings of the European Conference on Computer Vision (ECCV) 686\u2013701 (2018)","DOI":"10.1007\/978-3-030-01246-5_42"},{"key":"1038_CR20","doi-asserted-by":"crossref","unstructured":"Sarafianos, N., Xu, X., Kakadiaris, I.A.: Adversarial representation learning for text-to-image matching. Proceedings of the IEEE\/CVF International Conference on Computer Vision 5814\u20135824 (2019)","DOI":"10.1109\/ICCV.2019.00591"},{"key":"1038_CR21","unstructured":"Zheng, Z.: et al. Dual-path convolutional image-text embeddings with instance loss. arXiv preprint arXiv:1711.05535 (2017)"},{"key":"1038_CR22","doi-asserted-by":"crossref","unstructured":"Huang, Y., Wu, Q., Song, C., Wang, L.: Learning semantic concepts and order for image and sentence matching. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition 6163\u20136171 (2018)","DOI":"10.1109\/CVPR.2018.00645"},{"key":"1038_CR23","doi-asserted-by":"crossref","unstructured":"Gu, J., Cai, J., Joty, S.R., Niu, L., Wang, G.: Look, imagine and match: Improving textual-visual cross-modal retrieval with generative models. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition 7181\u20137189 (2018)","DOI":"10.1109\/CVPR.2018.00750"},{"issue":"1","key":"1038_CR24","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3284750","volume":"15","author":"Y Peng","year":"2019","unstructured":"Peng, Y., Qi, J.: Cm-gans: cross-modal generative adversarial networks for common representation learning. ACM Transact Multimedia Comput Commun Appl (TOMM) 15(1), 1\u201324 (2019)","journal-title":"ACM Transact Multimedia Comput Commun Appl (TOMM)"},{"key":"1038_CR25","doi-asserted-by":"crossref","unstructured":"Hu, P., Zhen, L., Peng, D., Liu, P.: Scalable deep multimodal learning for cross-modal retrieval. Proceedings of the 42nd international ACM SIGIR conference on research and development in information retrieval 635\u2013644 (2019)","DOI":"10.1145\/3331184.3331213"},{"key":"1038_CR26","doi-asserted-by":"crossref","unstructured":"Eisenschtat, A., Wolf, L.: Linking image and text with 2-way nets. Proceedings of the IEEE conference on computer vision and pattern recognition 4601\u20134611 (2017)","DOI":"10.1109\/CVPR.2017.201"},{"key":"1038_CR27","doi-asserted-by":"publisher","unstructured":"Devlin, J., Chang, M.-W., Lee, K., Toutanova, K.: Bert: Pre-training of deep bidirectional transformers for language understanding. Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers) 4171\u20134186 (2019). https:\/\/doi.org\/10.18653\/v1\/N19-1423","DOI":"10.18653\/v1\/N19-1423"},{"key":"1038_CR28","doi-asserted-by":"crossref","unstructured":"Hong, W.: et al. Gilbert: Generative vision-language pre-training for image-text retrieval. Proceedings of the 44th International ACM SIGIR Conference on Research and Development in Information Retrieval 1379\u20131388 (2021)","DOI":"10.1145\/3404835.3462838"},{"key":"1038_CR29","first-page":"11336","volume":"34\u201307","author":"G Li","year":"2020","unstructured":"Li, G., Duan, N., Fang, Y., Gong, M., Jiang, D.: Unicoder-vl: a universal encoder for vision and language by cross-modal pre-training. Proc. AAAI Conf. Artif. Intell. 34\u201307, 11336\u201311344 (2020)","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"1038_CR30","first-page":"13","volume":"32","author":"J Lu","year":"2019","unstructured":"Lu, J., Batra, D., Parikh, D., Lee, S.: Vilbert: pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. Adv. Neural Inform. Proces. Syst. 32, 13\u201323 (2019)","journal-title":"Adv. Neural Inform. Proces. Syst."},{"key":"1038_CR31","unstructured":"Chen, Y.-C.: et al. Uniter: learning universal image-text representations. arXiv preprint arXiv: 1909.11740 (2019)"},{"key":"1038_CR32","doi-asserted-by":"crossref","unstructured":"Huang, H.: et al. Unicoder: a universal language encoder by pre-training with multiple cross-lingual tasks. arXiv preprint arXiv:1909.00964 (2019)","DOI":"10.18653\/v1\/D19-1252"},{"key":"1038_CR33","doi-asserted-by":"crossref","unstructured":"Hudson, D.A., Manning, C.D.: Gqa: a new dataset for real-world visual reasoning and compositional question answering. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition 6700\u20136709 (2019)","DOI":"10.1109\/CVPR.2019.00686"},{"key":"1038_CR34","doi-asserted-by":"crossref","unstructured":"Plummer, B.A., et al.: Flickr30k entities: collecting region-to-phrase correspondences for richer image-to-sentence models. Proceedings of the IEEE international conference on computer vision 2641\u20132649 (2015)","DOI":"10.1109\/ICCV.2015.303"},{"key":"1038_CR35","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., et al.: Microsoft coco: common objects in context. European conference on computer vision 740\u2013755 (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"1038_CR36","doi-asserted-by":"crossref","unstructured":"Sharma, P., Ding, N., Goodman, S., Soricut, R.: Conceptual captions: A cleaned, hypernymed, image alt-text dataset for automatic image captioning. Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) 2556\u20132565 (2018)","DOI":"10.18653\/v1\/P18-1238"},{"key":"1038_CR37","first-page":"1143","volume":"24","author":"V Ordonez","year":"2011","unstructured":"Ordonez, V., Kulkarni, G., Berg, T.: Im2text: describing images using 1 million captioned photographs. Adv. Neural Inform. Process. Syst. 24, 1143\u20131151 (2011)","journal-title":"Adv. Neural Inform. Process. Syst."},{"key":"1038_CR38","unstructured":"Kiros, R., Salakhutdinov, R. Zemel, R.S.: Unifying visual-semantic embeddings with multimodal neural language models. arXiv preprint arXiv:1411.2539 (2014)"},{"key":"1038_CR39","doi-asserted-by":"publisher","unstructured":"Liu, Y., Guo, Y., Bakker, E.M., Lew, M.S.: Learning a recurrent residual fusion network for multimodal matching. 2017 IEEE International Conference on Computer Vision (ICCV) 4127\u20134136 (2017). https:\/\/doi.org\/10.1109\/ICCV.2017.442","DOI":"10.1109\/ICCV.2017.442"},{"key":"1038_CR40","unstructured":"Vaswani, A.: et al. Attention is all you need. arXiv (2017)"},{"key":"1038_CR41","doi-asserted-by":"crossref","unstructured":"Qu, L., Liu, M., Cao, D., Nie, L., Tian, Q.: Context-aware multi-view summarization network for image-text matching. Proceedings of the 28th ACM International Conference on Multimedia 1047\u20131055 (2020)","DOI":"10.1145\/3394171.3413961"},{"key":"1038_CR42","doi-asserted-by":"crossref","unstructured":"Wang, S., Wang, R., Yao, Z., Shan, S., Chen, X.: Cross-modal scene graph matching for relationship-aware image-text retrieval. Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision 1508\u20131517 (2020)","DOI":"10.1109\/WACV45572.2020.9093614"},{"key":"1038_CR43","first-page":"139","volume":"27","author":"I Goodfellow","year":"2014","unstructured":"Goodfellow, I., et al.: Generative adversarial nets. Adv. Neural Inform. Process. Syst. 27, 139\u2013144 (2014)","journal-title":"Adv. Neural Inform. Process. Syst."},{"key":"1038_CR44","doi-asserted-by":"crossref","unstructured":"Anderson, P., et al.: Bottom-up and top-down attention for image captioning and visual question answering. Proceedings of the IEEE conference on computer vision and pattern recognition 6077\u20136086 (2018)","DOI":"10.1109\/CVPR.2018.00636"},{"key":"1038_CR45","first-page":"91","volume":"28","author":"S Ren","year":"2015","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster r-cnn: towards real-time object detection with region proposal networks. Adv. Neural. Inf. Process. Syst. 28, 91\u201399 (2015)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1038_CR46","unstructured":"Ba, J.L., Kiros, J.R., Hinton, G.E.: Layer normalization. arXiv preprint arXiv:1607.06450 (2016)"},{"key":"1038_CR47","unstructured":"Hendrycks, D., Gimpel, K.: Gaussian error linear s (gelus). arXiv preprint arXiv: 1606.08415 (2016)"},{"key":"1038_CR48","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)"},{"key":"1038_CR49","doi-asserted-by":"crossref","unstructured":"Karpathy, A., Fei-Fei, L.: Deep visual-semantic alignments for generating image descriptions. Proceedings of the IEEE conference on computer vision and pattern recognition 3128\u20133137 (2015)","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"1038_CR50","doi-asserted-by":"crossref","unstructured":"Lee, K.-H., Chen, X., Hua, G., Hu, H., He, X.: Stacked cross attention for image-text matching. Proceedings of the European Conference on Computer Vision (ECCV) 201\u2013216 (2018)","DOI":"10.1007\/978-3-030-01225-0_13"},{"key":"1038_CR51","unstructured":"Collobert, R., Kavukcuoglu, K., Farabet, C.: Torch7: a matlab-like environment for machine learning. BigLearn NIPS Workshop (2011)"},{"key":"1038_CR52","doi-asserted-by":"crossref","unstructured":"Wehrmann, J., Lopes, M.A., More, M.D., Barros, R.C.: Fast self-attentive multimodal retrieval. 2018 IEEE Winter Conference on Applications of Computer Vision (WACV) 1871\u20131878 (2018)","DOI":"10.1109\/WACV.2018.00207"},{"key":"1038_CR53","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014)"},{"key":"1038_CR54","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. Proceedings of the IEEE conference on computer vision and pattern recognition 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"issue":"1","key":"1038_CR55","first-page":"3221","volume":"15","author":"L Van Der Maaten","year":"2014","unstructured":"Van Der Maaten, L.: Accelerating t-sne using tree-based algorithms. J. Mach. Learn. Res. 15(1), 3221\u20133245 (2014)","journal-title":"J. Mach. Learn. Res."}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-022-01038-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-022-01038-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-022-01038-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,5,30]],"date-time":"2023-05-30T14:12:50Z","timestamp":1685455970000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-022-01038-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,1,7]]},"references-count":55,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2023,6]]}},"alternative-id":["1038"],"URL":"https:\/\/doi.org\/10.1007\/s00530-022-01038-x","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"type":"print","value":"0942-4962"},{"type":"electronic","value":"1432-1882"}],"subject":[],"published":{"date-parts":[[2023,1,7]]},"assertion":[{"value":"25 September 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 December 2022","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 January 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}