{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,26]],"date-time":"2026-02-26T15:35:24Z","timestamp":1772120124712,"version":"3.50.1"},"reference-count":66,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2024,1,20]],"date-time":"2024-01-20T00:00:00Z","timestamp":1705708800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,20]],"date-time":"2024-01-20T00:00:00Z","timestamp":1705708800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"Natural Science Foundation of China","doi-asserted-by":"crossref","award":["U1936116"],"award-info":[{"award-number":["U1936116"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"Natural Science Foundation of China","doi-asserted-by":"crossref","award":["U1936116"],"award-info":[{"award-number":["U1936116"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100020084","name":"Guangzhou Municipal Science and Technology Bureau","doi-asserted-by":"publisher","award":["202102010412"],"award-info":[{"award-number":["202102010412"]}],"id":[{"id":"10.13039\/501100020084","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100020084","name":"Guangzhou Municipal Science and Technology Bureau","doi-asserted-by":"publisher","award":["202102010412"],"award-info":[{"award-number":["202102010412"]}],"id":[{"id":"10.13039\/501100020084","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Multimed Info Retr"],"published-print":{"date-parts":[[2024,3]]},"DOI":"10.1007\/s13735-023-00316-2","type":"journal-article","created":{"date-parts":[[2024,1,20]],"date-time":"2024-01-20T05:02:03Z","timestamp":1705726923000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["Cross-modal retrieval based on shared proxies"],"prefix":"10.1007","volume":"13","author":[{"given":"Yuxin","family":"Wei","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ligang","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guoping","family":"Qiu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guocan","family":"Cai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,1,20]]},"reference":[{"key":"316_CR1","doi-asserted-by":"crossref","unstructured":"Arya D, Rudinac S, Worring M (2019) HyperLearn: A distributed approach for representation learning in datasets with many modalities. In: Proceedings of the 27th ACM International Conference on Multimedia, MM 2019, pp 2245\u20132253. ACM, Nice, France","DOI":"10.1145\/3343031.3350572"},{"issue":"12","key":"316_CR2","doi-asserted-by":"publisher","first-page":"25236","DOI":"10.1109\/TITS.2022.3213320","volume":"23","author":"R Lan","year":"2022","unstructured":"Lan R, Tan Y, Wang X, Liu Z, Luo X (2022) Label guided discrete hashing for cross-modal retrieval. IEEE Trans Intell Transp Syst 23(12):25236\u201325248","journal-title":"IEEE Trans Intell Transp Syst"},{"key":"316_CR3","doi-asserted-by":"crossref","unstructured":"Cheng Q, Tan Z, Wen K, Chen C, Gu X (2023) Semantic pre-alignment and ranking learning with unified framework for cross-modal retrieval","DOI":"10.1109\/TCSVT.2022.3182549"},{"key":"316_CR4","doi-asserted-by":"publisher","unstructured":"Wang B, Yang Y, Xu X, Hanjalic A, Shen HT (2017) Adversarial cross-modal retrieval. In: Liu Q, Lienhart R, Wang H, Chen SK, Boll S, Chen YP, Friedland G, Li J, Yan S (eds) Proceedings of the 2017 ACM on Multimedia Conference, MM, pp 154\u2013162. ACM, Mountain View. https:\/\/doi.org\/10.1145\/3123266.3123326","DOI":"10.1145\/3123266.3123326"},{"issue":"2","key":"316_CR5","doi-asserted-by":"publisher","first-page":"405","DOI":"10.1109\/TMM.2017.2742704","volume":"20","author":"P Yuxin","year":"2017","unstructured":"Yuxin P, Jinwei Q, Xin H, Yuxin Y (2017) CCL: cross-modal correlation learning with multigrained fusion by hierarchical network. IEEE Trans Multimedia 20(2):405\u2013420. https:\/\/doi.org\/10.1109\/TMM.2017.2742704","journal-title":"IEEE Trans Multimedia"},{"key":"316_CR6","doi-asserted-by":"publisher","unstructured":"Zhen L, Hu P, Wang X, Peng D (2019) Deep supervised cross-modal retrieval. In: IEEE conference on computer vision and pattern recognition, CVPR 2019, pp 10394\u201310403. Computer vision foundation \/ IEEE, Long Beach, CA, USA. https:\/\/doi.org\/10.1109\/CVPR.2019.01064","DOI":"10.1109\/CVPR.2019.01064"},{"issue":"1s","key":"316_CR7","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3412847","volume":"17","author":"Z Chengyuan","year":"2021","unstructured":"Chengyuan Z, Jiayu S, Xiaofeng Z, Lei Z, Shichao Z (2021) HCMSL: hybrid cross-modal similarity learning for cross-modal retrieval. ACM Trans Multimed Comput Commun Appl (TOMM) 17(1s):1\u201322. https:\/\/doi.org\/10.1145\/3412847","journal-title":"ACM Trans Multimed Comput Commun Appl (TOMM)"},{"key":"316_CR8","doi-asserted-by":"publisher","unstructured":"Jing L, Vahdani E, Tan J, Tian Y (2021) Cross-modal center loss for 3d cross-modal retrieval. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp 3142\u20133151. https:\/\/doi.org\/10.1109\/cvpr46437.2021.00316","DOI":"10.1109\/cvpr46437.2021.00316"},{"issue":"3","key":"316_CR9","doi-asserted-by":"publisher","first-page":"1047","DOI":"10.1109\/TCYB.2018.2879846","volume":"50","author":"X Huang","year":"2020","unstructured":"Huang X, Peng Y, Yuan M (2020) MHTN: modal-adversarial hybrid transfer network for cross-modal retrieval. IEEE Trans. Cybern. 50(3):1047\u20131059","journal-title":"IEEE Trans. Cybern."},{"key":"316_CR10","doi-asserted-by":"publisher","unstructured":"Harold H (1936) Relations between two sets of variates. Breakthroughs in statistics: methodology and distribution. 162\u2013190. https:\/\/doi.org\/10.1093\/biomet\/28.3-4.321","DOI":"10.1093\/biomet\/28.3-4.321"},{"key":"316_CR11","doi-asserted-by":"publisher","unstructured":"Sharma A, Kumar A, Daum\u00e9III H, Jacobs DW (2012) Generalized multiview analysis: A discriminative latent space. In: 2012 IEEE conference on computer vision and pattern recognition, pp 2160\u20132167. IEEE Computer Society, Providence, RI, USA. https:\/\/doi.org\/10.1109\/cvpr.2012.6247923","DOI":"10.1109\/cvpr.2012.6247923"},{"issue":"2","key":"316_CR12","doi-asserted-by":"publisher","first-page":"210","DOI":"10.1007\/s11263-013-0658-4","volume":"106","author":"Y Gong","year":"2014","unstructured":"Gong Y, Ke Q, Isard M, Lazebnik S (2014) A multi-view embedding space for modeling internet images, tags, and their semantics. Int J Comput Vis 106(2):210\u2013233","journal-title":"Int J Comput Vis"},{"key":"316_CR13","doi-asserted-by":"publisher","first-page":"10","DOI":"10.1016\/j.neucom.2012.03.033","volume":"119","author":"H Zhang","year":"2013","unstructured":"Zhang H, Liu Y, Ma Z (2013) Fusing inherent and external knowledge with nonlinear learning for cross-media retrieval. Neurocomputing 119:10\u201316. https:\/\/doi.org\/10.1016\/j.neucom.2012.03.033","journal-title":"Neurocomputing"},{"key":"316_CR14","doi-asserted-by":"crossref","unstructured":"Yuan L, Wang T, Zhang X, Tay FEH, Jie Z, Liu W, Feng J (2020) Central similarity quantization for efficient image and video retrieval. In: 2020 IEEE\/CVF conference on computer vision and pattern recognition. CVPR 2020. Computer Vision Foundation\/IEEE, Seattle. pp 3083\u20133092","DOI":"10.1109\/CVPR42600.2020.00315"},{"key":"316_CR15","doi-asserted-by":"publisher","unstructured":"Chen Y, Lai Z, Ding Y, Lin K, Wong WK (2019) Deep supervised hashing with anchor graph. In: 2019 IEEE\/CVF international conference on computer vision, ICCV 2019, pp 9796\u20139804. IEEE, Seoul, Korea (South). https:\/\/doi.org\/10.1109\/iccv.2019.00989","DOI":"10.1109\/iccv.2019.00989"},{"key":"316_CR16","doi-asserted-by":"crossref","unstructured":"Carvalho M, Cad\u00e8ne R, Picard D, Soulier L, Thome N, Cord M (2018) Cross-modal retrieval in the cooking context: Learning semantic text-image embeddings. In: The 41st international ACM SIGIR conference on research & development in information retrieval. SIGIR 2018. ACM, Ann Arbor, pp 35\u201344","DOI":"10.1145\/3209978.3210036"},{"key":"316_CR17","doi-asserted-by":"crossref","unstructured":"Wang X, Han X, Huang W, Dong D, Scott MR (2019) Multi-similarity loss with general pair weighting for deep metric learning. In: IEEE conference on computer vision and pattern recognition. CVPR 2019. IEEE, Long Beach, pp 5022\u20135030","DOI":"10.1109\/CVPR.2019.00516"},{"key":"316_CR18","unstructured":"Goldberger J, Roweis ST, Hinton GE, Salakhutdinov R (2004) Neighbourhood components analysis. In: Advances in neural information processing systems 17 [Neural Information Processing Systems, NIPS 2004, Vancouver, British Columbia, pp 513\u2013520"},{"key":"316_CR19","doi-asserted-by":"publisher","unstructured":"Lin Z, Ding G, Hu M, Wang J (2015) Semantics-preserving hashing for cross-view retrieval. In: IEEE conference on computer vision and pattern recognition, CVPR 2015, pp 3864\u20133872. IEEE Computer Society, Boston. https:\/\/doi.org\/10.1109\/CVPR.2015.7299011","DOI":"10.1109\/CVPR.2015.7299011"},{"key":"316_CR20","doi-asserted-by":"publisher","unstructured":"Jiang Q, Li W (2017) Deep cross-modal hashing. In: 2017 IEEE conference on computer vision and pattern recognition, CVPR 2017, pp 3270\u20133278. IEEE Computer Society, Honolulu. https:\/\/doi.org\/10.1109\/CVPR.2017.348","DOI":"10.1109\/CVPR.2017.348"},{"key":"316_CR21","doi-asserted-by":"publisher","first-page":"113","DOI":"10.1016\/j.neucom.2020.02.043","volume":"396","author":"Q Lin","year":"2020","unstructured":"Lin Q, Cao W, He Z, He Z (2020) Semantic deep cross-modal hashing. Neurocomputing 396:113\u2013122. https:\/\/doi.org\/10.1016\/j.neucom.2020.02.043","journal-title":"Neurocomputing"},{"key":"316_CR22","doi-asserted-by":"publisher","unstructured":"Su S, Zhong Z, Zhang C (2019) Deep joint-semantics reconstructing hashing for large-scale unsupervised cross-modal retrieval. In: 2019 IEEE\/CVF international conference on computer vision, ICCV 2019, pp 3027\u20133035. IEEE, Seoul (South). https:\/\/doi.org\/10.1109\/ICCV.2019.00312","DOI":"10.1109\/ICCV.2019.00312"},{"key":"316_CR23","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2020.107335","volume":"104","author":"F Wu","year":"2020","unstructured":"Wu F, Jing X, Wu Z, Ji Y, Dong X, Luo X, Huang Q, Wang R (2020) Modality-specific and shared generative adversarial network for cross-modal retrieval. Pattern Recogn 104:107335. https:\/\/doi.org\/10.1016\/j.patcog.2020.107335","journal-title":"Pattern Recogn"},{"key":"316_CR24","doi-asserted-by":"publisher","unstructured":"Ranjan V, Rasiwasia N, Jawahar CV (2015) Multi-label cross-modal retrieval. In: 2015 IEEE international conference on computer vision, ICCV 2015, pp 4094\u20134102. IEEE Computer Society, Santiago. https:\/\/doi.org\/10.1109\/iccv.2015.466","DOI":"10.1109\/iccv.2015.466"},{"key":"316_CR25","unstructured":"Rasiwasia N, Mahajan D, Mahadevan V, Aggarwal G (2014) Cluster canonical correlation analysis. In: Proceedings of the seventeenth international conference on artificial intelligence and statistics, vol. 33. Reykjavik, Iceland, pp 823\u2013831. PMLR"},{"key":"316_CR26","unstructured":"Andrew G, Arora R, Bilmes JA, Livescu K (2013) Deep canonical correlation analysis. In: Proceedings of the 30th international conference on machine learning. JMLR workshop and conference proceedings, vol. 28, pp 1247\u20131255. Atlanta"},{"key":"316_CR27","unstructured":"Srivastava N, Salakhutdinov R (2012) Multimodal learning with deep boltzmann machines. In: Advances in neural information processing systems 25: 26th annual conference on neural information processing systems 2012, Lake Tahoe, pp 2231\u20132239. Citeseer"},{"key":"316_CR28","doi-asserted-by":"publisher","unstructured":"Fangxiang F, Xiaojie W, Ruifan L (2014) Cross-modal retrieval with correspondence autoencoder. In: Proceedings of the ACM international conference on multimedia, MM, pp 7\u201316. ACM, Orlando. https:\/\/doi.org\/10.1145\/2647868.2654902","DOI":"10.1145\/2647868.2654902"},{"key":"316_CR29","unstructured":"Wang W, Arora R, Livescu K, Bilmes JA (2015) On deep multi-view representation learning. In: Proceedings of the 32nd international conference on machine learning, ICML 2015. JMLR workshop and conference proceedings, vol. 37, pp 1083\u20131092. JMLR.org, Lille"},{"key":"316_CR30","doi-asserted-by":"publisher","unstructured":"Wang C, Yang H, Meinel C (2015) Deep semantic mapping for cross-modal retrieval. In: 27th IEEE international conference on tools with artificial intelligence, ICTAI 2015, pp 234\u2013241. IEEE Computer Society, Vietri sul Mare. https:\/\/doi.org\/10.1109\/ictai.2015.45","DOI":"10.1109\/ictai.2015.45"},{"key":"316_CR31","doi-asserted-by":"publisher","unstructured":"Li Z, Lu W, Bao E, Xing W (2015) Learning a semantic space by deep network for cross-media retrieval. In: The 21st international conference on distributed multimedia systems, pp 199\u2013203. Knowledge Systems Institute, Vancouver. https:\/\/doi.org\/10.18293\/dms2015-005","DOI":"10.18293\/dms2015-005"},{"issue":"2","key":"316_CR32","doi-asserted-by":"publisher","first-page":"449","DOI":"10.1109\/tcyb.2016.2519449","volume":"47","author":"Y Wei","year":"2017","unstructured":"Wei Y, Zhao Y, Lu C, Wei S, Liu L, Zhu Z, Yan S (2017) Cross-modal retrieval with CNN visual features: A new baseline. IEEE Trans. Cybern. 47(2):449\u2013460. https:\/\/doi.org\/10.1109\/tcyb.2016.2519449","journal-title":"IEEE Trans. Cybern."},{"key":"316_CR33","doi-asserted-by":"publisher","unstructured":"Castrej\u00f3n L, Aytar Y, Vondrick C, Pirsiavash H, Torralba A (2016) Learning aligned cross-modal representations from weakly aligned data. In: 2016 IEEE conference on computer vision and pattern recognition, CVPR 2016, pp 2940\u20132949. IEEE Computer Society, Las Vegas. https:\/\/doi.org\/10.1109\/cvpr.2016.321","DOI":"10.1109\/cvpr.2016.321"},{"key":"316_CR34","doi-asserted-by":"publisher","unstructured":"Goodfellow IJ, Pouget-Abadie J, Mirza M, Xu B, Warde-Farley D, Ozair S, Courville AC, Bengio Y (2014) Generative adversarial nets. In: Ghahramani Z, Welling M, Cortes C, Lawrence ND, Weinberger KQ (eds) Advances in neural information processing systems 27: annual conference on neural information processing systems 2014, Montreal, Quebec, pp 2672\u20132680. https:\/\/doi.org\/10.3156\/jsoft.29.5_177_2","DOI":"10.3156\/jsoft.29.5_177_2"},{"issue":"1","key":"316_CR35","doi-asserted-by":"publisher","first-page":"22","DOI":"10.1145\/3284750","volume":"15","author":"Y Peng","year":"2019","unstructured":"Peng Y, Qi J (2019) CM-GANs: cross-modal generative adversarial networks for common representation learning. ACM Trans. Multim. Comput. Commun. Appl. 15(1):22\u201312224. https:\/\/doi.org\/10.1145\/3284750","journal-title":"ACM Trans. Multim. Comput. Commun. Appl."},{"issue":"2","key":"316_CR36","doi-asserted-by":"publisher","first-page":"657","DOI":"10.1007\/s11280-018-0541-x","volume":"22","author":"X Xu","year":"2019","unstructured":"Xu X, He L, Lu H, Gao L, Ji Y (2019) Deep adversarial metric learning for cross-modal retrieval. World Wide Web 22(2):657\u2013672. https:\/\/doi.org\/10.1007\/s11280-018-0541-x","journal-title":"World Wide Web"},{"issue":"2","key":"316_CR37","doi-asserted-by":"publisher","first-page":"920","DOI":"10.1109\/TCSVT.2022.3203247","volume":"33","author":"L Liao","year":"2023","unstructured":"Liao L, Yang M, Zhang B (2023) Deep supervised dual cycle adversarial network for cross-modal retrieval. IEEE Trans Circ Syst Video Technol 33(2):920\u2013934. https:\/\/doi.org\/10.1109\/TCSVT.2022.3203247","journal-title":"IEEE Trans Circ Syst Video Technol"},{"key":"316_CR38","doi-asserted-by":"crossref","unstructured":"Kim S, Seo M, Laptev I, Cho M, Kwak S (2019) Deep metric learning beyond binary supervision. In: IEEE conference on computer vision and pattern recognition. CVPR 2019. Computer Vision Foundation\/IEEE, Long Beach, pp 2288\u20132297","DOI":"10.1109\/CVPR.2019.00239"},{"key":"316_CR39","unstructured":"Sohn K (2016) Improved deep metric learning with multi-class N-pair loss objective. In: Lee DD, Sugiyama M, von Luxburg U, Guyon I, Garnett R (eds) Advances in neural information processing systems 29: annual conference on neural information processing systems 2016, Barcelona, pp 1849\u20131857"},{"key":"316_CR40","doi-asserted-by":"publisher","first-page":"448","DOI":"10.1007\/978-3-030-58586-0_27","volume":"12369","author":"EW Teh","year":"2020","unstructured":"Teh EW, DeVries T, Taylor GW (2020) Proxynca++: revisiting and revitalizing proxy neighborhood component analysis. Springer International Publishing eBooks 12369:448\u2013464. https:\/\/doi.org\/10.1007\/978-3-030-58586-0_27","journal-title":"Springer International Publishing eBooks"},{"key":"316_CR41","doi-asserted-by":"crossref","unstructured":"Yang Z, Bastan M, Zhu X, Gray D, Samaras D (2022) Hierarchical proxy-based loss for deep metric learning. In: IEEE\/CVF winter conference on applications of computer vision. WACV 2022. IEEE, Waikoloa, pp 449\u2013458","DOI":"10.1109\/WACV51458.2022.00052"},{"key":"316_CR42","doi-asserted-by":"publisher","unstructured":"Qian Q, Shang L, Sun B, Hu J, Tacoma T, Li H, Jin R (2019) SoftTriple loss: deep metric learning without triplet sampling. In: 2019 IEEE\/CVF international conference on computer vision, ICCV 2019, pp 6449\u20136457. IEEE, Seoul, (South). https:\/\/doi.org\/10.1109\/iccv.2019.00655","DOI":"10.1109\/iccv.2019.00655"},{"key":"316_CR43","doi-asserted-by":"publisher","unstructured":"Aziere N, Todorovic S (2019) Ensemble deep manifold similarity learning using hard proxies. In: IEEE conference on computer vision and pattern recognition, CVPR 2019, pp 7299\u20137307. Computer Vision Foundation\/IEEE, Long Beach. https:\/\/doi.org\/10.1109\/cvpr.2019.00747","DOI":"10.1109\/cvpr.2019.00747"},{"key":"316_CR44","doi-asserted-by":"crossref","unstructured":"Movshovitz-Attias Y, Toshev A, Leung TK, Ioffe S, Singh S (2017) No fuss distance metric learning using proxies. In: IEEE international conference on computer vision. ICCV 2017. IEEE Computer Society, Venice, pp 360\u2013368","DOI":"10.1109\/ICCV.2017.47"},{"key":"316_CR45","doi-asserted-by":"publisher","unstructured":"Kim S, Kim D, Cho M, Kwak S (2020) Proxy anchor loss for deep metric learning. In: 2020 IEEE\/CVF conference on computer vision and pattern recognition, CVPR 2020, pp 3235\u20133244. Computer Vision Foundation\/IEEE, Seattle. https:\/\/doi.org\/10.1109\/cvpr42600.2020.00330","DOI":"10.1109\/cvpr42600.2020.00330"},{"key":"316_CR46","volume-title":"3rd international conference on learning representations, ICLR 2015","author":"K Simonyan","year":"2015","unstructured":"Simonyan K, Zisserman A (2015) Very deep convolutional networks for large-scale image recognition. In: Bengio Y, LeCun Y (eds) 3rd international conference on learning representations, ICLR 2015. San Diego, CA"},{"key":"316_CR47","doi-asserted-by":"crossref","unstructured":"Kim Y (2014) Convolutional neural networks for sentence classification. In: Proceedings of the 2014 conference on empirical methods in natural language processing, EMNLP 2014, pp 1746\u20131751. ACL, Doha, Qatar","DOI":"10.3115\/v1\/D14-1181"},{"key":"316_CR48","unstructured":"Mikolov T, Sutskever I, Chen K, Corrado GS, Dean J (2013) Distributed representations of words and phrases and their compositionality. In: Advances in neural information processing systems 26: 27th annual conference on neural information processing systems 2013. Proceedings of a meeting held December 5-8, 2013, Lake Tahoe, pp 3111\u20133119"},{"key":"316_CR49","unstructured":"Kingma DP, Ba J (2015) Adam: A method for stochastic optimization. In: Bengio Y, LeCun Y (eds) International conference on learning representations, ICLR 2015, San Diego. http:\/\/arxiv.org\/abs\/1412.6980"},{"issue":"3","key":"316_CR50","doi-asserted-by":"publisher","first-page":"521","DOI":"10.1109\/tpami.2013.142","volume":"36","author":"JC Pereira","year":"2014","unstructured":"Pereira JC, Coviello E, Doyle G, Rasiwasia N, Lanckriet GRG, Levy R, Vasconcelos N (2014) On the role of correlation and abstraction in cross-modal multimedia retrieval. IEEE Trans Pattern Anal Mach Intell 36(3):521\u2013535. https:\/\/doi.org\/10.1109\/tpami.2013.142","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"316_CR51","unstructured":"Rashtchian C, Young P, Hodosh M, Hockenmaier J (2010) Collecting image annotations using amazon\u2019s mechanical turk. In: Proceedings of the 2010 workshop on creating speech and language data with Amazon\u2019s mechanical turk, pp 139\u2013147. Association for Computational Linguistics, Los Angeles"},{"key":"316_CR52","doi-asserted-by":"publisher","unstructured":"Chua T, Tang J, Hong R, Li H, Luo Z, Zheng Y (2009) NUS-WIDE: A real-world web image database from national university of singapore. In: Proceedings of the 8th ACM international conference on image and video retrieval, CIVR 2009. ACM, Santorini Island, Greece. https:\/\/doi.org\/10.1145\/1646396.1646452","DOI":"10.1145\/1646396.1646452"},{"key":"316_CR53","doi-asserted-by":"crossref","unstructured":"Zhai X, Peng Y, Xiao J (2014) Learning cross-media joint representation with sparse and semisupervised regularization. In: IEEE transactions on circuits and systems for video technology","DOI":"10.1109\/TCSVT.2013.2276704"},{"key":"316_CR54","doi-asserted-by":"crossref","unstructured":"Lin T-Y, Maire M, Belongie S, Hays J, Perona P, Ramanan D, Doll\u00e1r P, Zitnick CL (2014) Microsoft coco: Common objects in context. In: Computer vision\u2013ECCV 2014: 13th European conference, Zurich, September 6-12, 2014, Proceedings, Part V 13, pp 740\u2013755. Springer","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"316_CR55","doi-asserted-by":"crossref","unstructured":"Karpathy A, Fei-Fei L (2015) Deep visual-semantic alignments for generating image descriptions. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3128\u20133137","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"316_CR56","unstructured":"Simonyan K, Zisserman A (2015) Very deep convolutional networks for large-scale image recognition. In: 3rd international conference on learning representations, ICLR 2015, San Diego"},{"issue":"5","key":"316_CR57","doi-asserted-by":"publisher","first-page":"923","DOI":"10.1109\/TMM.2007.900138","volume":"9","author":"N Rasiwasia","year":"2007","unstructured":"Rasiwasia N, Moreno PJ, Vasconcelos N (2007) Bridging the gap: query by semantic example. IEEE Trans Multim 9(5):923\u2013938. https:\/\/doi.org\/10.1109\/TMM.2007.900138","journal-title":"IEEE Trans Multim"},{"key":"316_CR58","doi-asserted-by":"publisher","unstructured":"Zhai X, Peng Y, Xiao J (2012) Cross-modality correlation propagation for cross-media retrieval. In: 2012 IEEE international conference on acoustics, speech and signal processing, ICASSP 2012, pp 2337\u20132340. IEEE, Kyoto. https:\/\/doi.org\/10.1109\/icassp.2012.6288383","DOI":"10.1109\/icassp.2012.6288383"},{"key":"316_CR59","unstructured":"Peng Y, Peng Y, Xiao J (2013) Heterogeneous metric learning with joint graph regularization for cross-media retrieval. In: Proceedings of the Twenty-Seventh AAAI conference on artificial intelligence. AAAI Press, Bellevue, Washington"},{"issue":"6","key":"316_CR60","doi-asserted-by":"publisher","first-page":"965","DOI":"10.1109\/tcsvt.2013.2276704","volume":"24","author":"X Zhai","year":"2014","unstructured":"Zhai X, Peng Y, Xiao J (2014) Learning cross-media joint representation with sparse and semisupervised regularization. IEEE Trans Circ Syst Video Technol 24(6):965\u2013978. https:\/\/doi.org\/10.1109\/tcsvt.2013.2276704","journal-title":"IEEE Trans Circ Syst Video Technol"},{"key":"316_CR61","unstructured":"Peng Y, Huang X, Qi J (2016) Cross-media shared representation by hierarchical learning with multiple deep networks. In: Proceedings of the Twenty-Fifth international joint conference on artificial intelligence, IJCAI 2016, pp 3846\u20133853. IJCAI\/AAAI Press, New York"},{"key":"316_CR62","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2022.3182549","author":"Q Cheng","year":"2022","unstructured":"Cheng Q, Tan Z, Wen K, Chen C, Gu X (2022) Semantic pre-alignment and ranking learning with unified framework for cross-modal retrieval. IEEE Trans Circ Syst Video Technol. https:\/\/doi.org\/10.1109\/TCSVT.2022.3182549","journal-title":"IEEE Trans Circ Syst Video Technol"},{"key":"316_CR63","doi-asserted-by":"publisher","unstructured":"Hu P, Zhen L, Peng D, Liu P (2019) Scalable deep multimodal learning for cross-modal retrieval. In: Proceedings of the 42nd international ACM SIGIR conference on research and development in information retrieval, SIGIR 2019, pp 635\u2013644. ACM, Paris. https:\/\/doi.org\/10.1145\/3331184.3331213","DOI":"10.1145\/3331184.3331213"},{"key":"316_CR64","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. Advances in neural information processing systems 30"},{"key":"316_CR65","unstructured":"Radford A, Kim JW, Hallacy C, Ramesh A, Goh G, Agarwal S, Sastry G, Askell A, Mishkin P, Clark J, et al (2021) Learning transferable visual models from natural language supervision. In: International conference on machine learning, pp 8748\u20138763. PMLR"},{"key":"316_CR66","unstructured":"Van\u00a0der Maaten L, Hinton G (2008) Visualizing data using t-SNE. J Mach Learn Res 9(11)"}],"container-title":["International Journal of Multimedia Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-023-00316-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13735-023-00316-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-023-00316-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,28]],"date-time":"2024-03-28T07:11:40Z","timestamp":1711609900000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13735-023-00316-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,1,20]]},"references-count":66,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2024,3]]}},"alternative-id":["316"],"URL":"https:\/\/doi.org\/10.1007\/s13735-023-00316-2","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-2667484\/v1","asserted-by":"object"}]},"ISSN":["2192-6611","2192-662X"],"issn-type":[{"value":"2192-6611","type":"print"},{"value":"2192-662X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,1,20]]},"assertion":[{"value":"8 March 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 November 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 December 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 January 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"5"}}