{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,10]],"date-time":"2026-06-10T11:40:45Z","timestamp":1781091645542,"version":"3.54.1"},"reference-count":52,"publisher":"Springer Science and Business Media LLC","issue":"21-22","license":[{"start":{"date-parts":[[2019,2,21]],"date-time":"2019-02-21T00:00:00Z","timestamp":1550707200000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2019,2,21]],"date-time":"2019-02-21T00:00:00Z","timestamp":1550707200000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/100015570","name":"NFSC","doi-asserted-by":"crossref","award":["61762021"],"award-info":[{"award-number":["61762021"]}],"id":[{"id":"10.13039\/100015570","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2020,6]]},"DOI":"10.1007\/s11042-019-7343-8","type":"journal-article","created":{"date-parts":[[2019,2,21]],"date-time":"2019-02-21T00:04:28Z","timestamp":1550707468000},"page":"14733-14750","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["Semantic consistent adversarial cross-modal retrieval exploiting semantic similarity"],"prefix":"10.1007","volume":"79","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5241-7703","authenticated-orcid":false,"given":"Weihua","family":"Ou","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ruisheng","family":"Xuan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jianping","family":"Gou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Quan","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yongfeng","family":"Cao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2019,2,21]]},"reference":[{"key":"7343_CR1","unstructured":"Andrew G, Arora R, Bilmes J, Livescu K (2013) Deep canonical correlation analysis. In: The 30th international conference on machine learning (ICML), pp 1247\u20131255"},{"key":"7343_CR2","doi-asserted-by":"crossref","unstructured":"Chua TS, Tang J, Hong R, Li H, Luo Z, Zheng Y (2009) Nus-wide: a real-world web image database from national university of singapore. In: ACM International conference on image and video retrieval, pp 48","DOI":"10.1145\/1646396.1646452"},{"issue":"3","key":"7343_CR3","doi-asserted-by":"publisher","first-page":"521","DOI":"10.1109\/TPAMI.2013.142","volume":"36","author":"PJ Costa","year":"2014","unstructured":"Costa PJ, Coviello E, Doyle G, Rasiwasia N, Lanckriet GR, Levy R, Vasconcelos N (2014) On the role of correlation and abstraction in cross-modal multimedia retrieval. IEEE Trans Pattern Anal Mach Intell 36(3):521\u201335","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"issue":"8","key":"7343_CR4","doi-asserted-by":"publisher","first-page":"3893","DOI":"10.1109\/TIP.2018.2821921","volume":"27","author":"C Deng","year":"2018","unstructured":"Deng C, Chen Z, Liu X, Gao X, Tao D (2018) Triplet-based deep hashing network for cross-modal retrieval. IEEE Trans Image Process 27(8):3893\u20133903","journal-title":"IEEE Trans Image Process"},{"key":"7343_CR5","unstructured":"Dong S, Gao Z, Sun S, Wang X, Li M, Zhang H, Yang G, Liu H, Li S (2018) Holistic and deep feature pyramids for saliency detection. In: British machine vision conference (BMVC), Northumbria University, Newcastle, UK, September 3\u20136, p 67"},{"key":"7343_CR6","doi-asserted-by":"crossref","unstructured":"Feng F, Wang X, Li R (2014) Cross-modal retrieval with correspondence autoencoder. The 22nd International conference on multimedia (ACM):7\u201316","DOI":"10.1145\/2647868.2654902"},{"issue":"9","key":"7343_CR7","doi-asserted-by":"publisher","first-page":"2045","DOI":"10.1109\/TMM.2017.2729019","volume":"19","author":"L Gao","year":"2017","unstructured":"Gao L, Guo Z, Zhang H, Xu X, Shen HT (2017) Video captioning with attention-based lstm and semantic consistency. IEEE Trans Multimedia 19(9):2045\u20132055","journal-title":"IEEE Trans Multimedia"},{"issue":"1","key":"7343_CR8","doi-asserted-by":"publisher","first-page":"273","DOI":"10.1109\/TMI.2017.2746879","volume":"37","author":"Z Gao","year":"2018","unstructured":"Gao Z, Li Y, Sun Y, Yang J, Xiong H, Zhang H, Liu X, Wu W, Liang D, Li S (2018) Motion tracking of the carotid artery wall from ultrasound image sequences: a nonlinear state-space approach. IEEE Trans Med Imaging 37(1):273\u2013283","journal-title":"IEEE Trans Med Imaging"},{"key":"7343_CR9","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.media.2017.01.004","volume":"37","author":"Z Gao","year":"2017","unstructured":"Gao Z, Xiong H, Liu X, Zhang H, Ghista D, Wu W, Li S (2017) Robust estimation of carotid artery wall motion using the elasticity-based state-space approach. Med Image Anal 37:1\u201321","journal-title":"Med Image Anal"},{"issue":"2","key":"7343_CR10","doi-asserted-by":"publisher","first-page":"210","DOI":"10.1007\/s11263-013-0658-4","volume":"106","author":"Y Gong","year":"2014","unstructured":"Gong Y, Ke Q, Isard M, Lazebnik S (2014) A multi-view embedding space for modeling internet images, tags, and their semantics. Int J Comput Vis 106 (2):210\u2013233","journal-title":"Int J Comput Vis"},{"key":"7343_CR11","unstructured":"Gong M, Zhang K, Liu T, Tao D, Glymour C, Sch\u00f6lkopf B (2016) Domain adaptation with conditional transferable components. In: Proceedings of the 33nd international conference on machine learning (ICML), New York City, NY, USA, June 19\u201324, vol 48, pp 2839\u20132848"},{"key":"7343_CR12","unstructured":"Goodfellow I, Pouget-Abadie J, Mirza M, Xu B, Warde-Farley D, Ozair S, Courville A, Bengio Y (2014) Generative adversarial nets. In: Advances in neural information processing systems (NIPS), pp 2672\u20132680"},{"issue":"12","key":"7343_CR13","doi-asserted-by":"publisher","first-page":"2639","DOI":"10.1162\/0899766042321814","volume":"16","author":"DR Hardoon","year":"2004","unstructured":"Hardoon DR, Szedmak S, Shawetaylor J (2004) Canonical correlation analysis: an overview with application to learning methods. Neural Comput 16(12):2639\u20132664","journal-title":"Neural Comput"},{"issue":"8","key":"7343_CR14","doi-asserted-by":"publisher","first-page":"3698","DOI":"10.1109\/TIP.2016.2570553","volume":"25","author":"Z He","year":"2016","unstructured":"He Z, Li X, You X, Tao D, Tang YY (2016) Connected component model for multi-object tracking. IEEE Trans Image Process 25(8):3698\u20133711","journal-title":"IEEE Trans Image Process"},{"key":"7343_CR15","unstructured":"Hua Y, Tian H, Cai A, Shi P (2016) Cross-modal correlation learning with deep convolutional architecture. In: Visual communications and image processing, pp 1\u20134"},{"key":"7343_CR16","doi-asserted-by":"publisher","unstructured":"Huang X, Peng Y, Yuan M (2018) Mhtn: Modal-adversarial hybrid transfer network for cross-modal retrieval. IEEE Transactions on Cybernetics, \nhttps:\/\/doi.org\/10.1109\/TCYB.2018.2879846","DOI":"10.1109\/TCYB.2018.2879846"},{"key":"7343_CR17","unstructured":"Jacobs DW, Daume H, Kumar A, Sharma A (2012) Generalized multiview analysis: a discriminative latent space. In: IEEE conference on computer vision and pattern recognition (CVPR), pp 2160\u2013 2167"},{"key":"7343_CR18","doi-asserted-by":"crossref","unstructured":"Jiang X, Wu F, Li X, Zhao Z, Lu W, Tang S, Zhuang Y (2015) Deep compositional cross-modal learning to rank via local-global alignment. In: International conference on multimedia ACM, pp 69\u201378","DOI":"10.1145\/2733373.2806240"},{"issue":"3","key":"7343_CR19","doi-asserted-by":"publisher","first-page":"370","DOI":"10.1109\/TMM.2015.2390499","volume":"17","author":"C Kang","year":"2015","unstructured":"Kang C, Xiang S, Liao S, Xu C, Pan C (2015) Learning consistent feature representation for cross-modal multimedia retrieval. IEEE Trans Multimedia 17 (3):370\u2013381","journal-title":"IEEE Trans Multimedia"},{"key":"7343_CR20","doi-asserted-by":"crossref","unstructured":"Li C, Deng C, Li N, Liu W, Gao X, Tao D (2018) Self-supervised adversarial hashing networks for cross-modal retrieval. arXiv:\n1804.01223","DOI":"10.1109\/CVPR.2018.00446"},{"key":"7343_CR21","unstructured":"Li H, Xu X, Lu H, Yang Y, Shen F, Shen HT (2017) Unsupervised cross-modal retrieval through adversarial learning. In: IEEE international conference on multimedia and expo, pp 1153\u20131158"},{"key":"7343_CR22","doi-asserted-by":"publisher","first-page":"189","DOI":"10.1016\/j.knosys.2017.07.032","volume":"134","author":"Q Liu","year":"2017","unstructured":"Liu Q, Lu X, He Z, Zhang C, Wen-sheng C (2017) Deep convolutional neural networks for thermal infrared object tracking. Knowl-Based Syst 134:189\u2013198","journal-title":"Knowl-Based Syst"},{"key":"7343_CR23","doi-asserted-by":"publisher","unstructured":"Lu H, Li B, Zhu J, Li Y, Li Y, Xu X, He L, Li X, Li J, Serikawa S (2017) Wound intensity correction and segmentation with convolutional neural networks. Concurrency & Computation Practice & Experience. \nhttps:\/\/doi.org\/10.1002\/cpe.3927","DOI":"10.1002\/cpe.3927"},{"issue":"2","key":"7343_CR24","doi-asserted-by":"publisher","first-page":"368","DOI":"10.1007\/s11036-017-0932-8","volume":"23","author":"H Lu","year":"2018","unstructured":"Lu H, Li Y, Chen M, Kim H, Serikawa S (2018) Brain intelligence: Go beyond artificial intelligence. Mobile Networks & Applications 23(2):368\u2013375","journal-title":"Mobile Networks & Applications"},{"key":"7343_CR25","doi-asserted-by":"publisher","unstructured":"Lu H, Li Y, Mu S, Wang D, Kim H, Serikawa S Motor anomaly detection for unmanned aerial vehicles using reinforcement learning. IEEE Internet Things J, \nhttps:\/\/doi.org\/10.1109\/JIOT.2017.2737479","DOI":"10.1109\/JIOT.2017.2737479"},{"key":"7343_CR26","unstructured":"Lu H, Li Y, Uemura T, Ge Z, Xu X, Li H, Serikawa S, Kim H (2017) Fdcnet: filtering deep convolutional network for marine organism classification. Multimed Tools Appl(2):1\u201314"},{"key":"7343_CR27","doi-asserted-by":"publisher","unstructured":"Lu H, Li Y, Uemura T, Kim H, Serikawa S (2018) Low illumination underwater light field images reconstruction using deep convolutional neural networks. Futur Gener Comput Syst. \nhttps:\/\/doi.org\/10.1016\/j.future.2018.01.001","DOI":"10.1016\/j.future.2018.01.001"},{"issue":"11","key":"7343_CR28","first-page":"2579","volume":"9","author":"Laurensvander Maaten","year":"2008","unstructured":"Maaten Laurens van der, Hinton G (2008) Visualizing data using t-sne. J Mach Learn Res 9(11):2579\u20132605","journal-title":"J Mach Learn Res"},{"key":"7343_CR29","unstructured":"Ngiam J, Khosla A, Kim M, Nam J, Lee H, Ng AY (2011) Multimodal deep learning. In: The 28th international conference on machine learning (ICML), Washington, USA, from June 28 to July 2, 2011, pp 689\u2013696"},{"key":"7343_CR30","unstructured":"Peng Y, Huang X, Zhao Y (2017) An overview of cross-media retrieval: Concepts, methodologies, benchmarks and challenges. IEEE Trans Circuits Syst Video Technol: 1\u201314"},{"key":"7343_CR31","unstructured":"Peng Y, Qi J, Yuan Y Cm-gans: Cross-modal generative adversarial networks for common representation learning. arXiv:\n1710.05106"},{"key":"7343_CR32","doi-asserted-by":"publisher","unstructured":"Peng Y, Zhang J, Yuan M (2018) Sch-gan: Semi-supervised cross-modal hashing by generative adversarial network. IEEE Transactions on Cybernetics, \nhttps:\/\/doi.org\/10.1109\/TCYB.2018.2868826","DOI":"10.1109\/TCYB.2018.2868826"},{"key":"7343_CR33","doi-asserted-by":"crossref","unstructured":"Rasiwasia N, Pereira JC, Coviello E, Doyle G, Lanckriet GRG, Levy R, Vasconcelos N (2010) A new approach to cross-modal multimedia retrieval. In: International conference on multimedia (ACM), pp 251\u2013260","DOI":"10.1145\/1873951.1873987"},{"key":"7343_CR34","first-page":"34","volume":"3940","author":"R Rosipal","year":"2006","unstructured":"Rosipal R, Kramer N (2006) Overview and recent advances in partial least squares. International Statistical and Optimization Perspectives Workshop 3940:34\u201351","journal-title":"International Statistical and Optimization Perspectives Workshop"},{"key":"7343_CR35","doi-asserted-by":"publisher","unstructured":"Song J, Yuyu G, Gao L, Li X, Hanjalic A, Shen HT (2018) From deterministic to generative: Multi-modal stochastic rnns for video captioning. IEEE Transactions on Neural Networks and Learning Systems, \nhttps:\/\/doi.org\/10.1109\/TNNLS.2018.2851077","DOI":"10.1109\/TNNLS.2018.2851077"},{"issue":"7","key":"7343_CR36","doi-asserted-by":"publisher","first-page":"3210","DOI":"10.1109\/TIP.2018.2814344","volume":"27","author":"J Song","year":"2018","unstructured":"Song J, Zhang H, Li X, Gao L, Wang M, Hong R (2018) Self-supervised video hashing with hierarchical binary auto-encoder. IEEE Trans Image Process 27 (7):3210","journal-title":"IEEE Trans Image Process"},{"key":"7343_CR37","unstructured":"Srivastava N, Salakhutdinov R (2012) Learning representations for multimodal data with deep belief nets. ICML workshop:79"},{"issue":"6","key":"7343_CR38","doi-asserted-by":"publisher","first-page":"1247","DOI":"10.1162\/089976600300015349","volume":"12","author":"JB Tenenbaum","year":"2000","unstructured":"Tenenbaum JB, Freeman WT (2000) Separating style and content with bilinear models. Neural Comput 12(6):1247\u20131283","journal-title":"Neural Comput"},{"key":"7343_CR39","doi-asserted-by":"crossref","unstructured":"Wang B, Yang Y, Xu X, Hanjalic A, Shen HT (2017) Adversarial cross-modal retrieval. In: International conference on multimedia (ACM), pp 154\u2013162","DOI":"10.1145\/3123266.3123326"},{"key":"7343_CR40","doi-asserted-by":"crossref","unstructured":"Wang K, He R, Wang W, Wang L, Tan T (2013) Learning coupled feature spaces for cross-modal matching. IEEE International Conference on Computer Vision (ICCV):2088\u20132095","DOI":"10.1109\/ICCV.2013.261"},{"key":"7343_CR41","doi-asserted-by":"crossref","unstructured":"Wang J, He Y, Kang C, Xiang S, Pan C (2015) Image-text cross-modal retrieval via modality-specific feature learning. In: International conference on multimedia retrieval (ACM), pp 347\u2013354","DOI":"10.1145\/2671188.2749341"},{"issue":"10","key":"7343_CR42","doi-asserted-by":"publisher","first-page":"2010","DOI":"10.1109\/TPAMI.2015.2505311","volume":"38","author":"K Wang","year":"2016","unstructured":"Wang K, He R, Wang L, Wang W, Tan T (2016) Joint feature selection and subspace learning for cross-modal retrieval. IEEE Trans Pattern Anal Mach Intell (PAMI) 38(10):2010\u20132023","journal-title":"IEEE Trans Pattern Anal Mach Intell (PAMI)"},{"key":"7343_CR43","unstructured":"Wang K, Yin Q, Wang W, Wu S, Wang L (2016) A comprehensive survey on cross-modal retrieval. arXiv:\n1607.06215"},{"issue":"2","key":"7343_CR44","first-page":"449","volume":"47","author":"Y Wei","year":"2016","unstructured":"Wei Y, Zhao Y, Lu C , Wei S, Liu L, Zhu Z, Yan S (2016) Cross-modal retrieval with cnn visual features: a new baseline. IEEE Transactions on Cybernetics 47(2):449\u2013460","journal-title":"IEEE Transactions on Cybernetics"},{"key":"7343_CR45","unstructured":"Xi Z, Zhou S, Feng J, Lai H, Li B, Pan Y, Yin J, Yan S (2017) Hashgan: Attention-aware deep adversarial hashing for cross modal retrieval. arXiv:\n1711.09347"},{"issue":"2","key":"7343_CR46","doi-asserted-by":"publisher","first-page":"208","DOI":"10.1109\/TMM.2015.2508146","volume":"18","author":"T Xu","year":"2016","unstructured":"Xu T, Yang Y, Deng C, Gao X (2016) Coupled dictionary learning with common label alignment for cross-modal retrieval. IEEE Trans Multimedia 18 (2):208\u2013218","journal-title":"IEEE Trans Multimedia"},{"key":"7343_CR47","doi-asserted-by":"publisher","first-page":"191","DOI":"10.1016\/j.neucom.2015.11.133","volume":"213","author":"X Xu","year":"2016","unstructured":"Xu X, Li H, Shimada A, Taniguchi RI, Huimin L (2016) Learning unified binary codes for cross-modal retrieval via latent semantic hashing. Neurocomputing 213:191\u2013203","journal-title":"Neurocomputing"},{"key":"7343_CR48","doi-asserted-by":"publisher","unstructured":"Xu X, Li H, Lu H, Gao L, Ji Y (2018) Deep adversarial metric learning for cross-modal retrieval. World Wide Web:1\u201316. \nhttps:\/\/doi.org\/10.1007\/s11280-018-0541-x","DOI":"10.1007\/s11280-018-0541-x"},{"issue":"5","key":"7343_CR49","doi-asserted-by":"publisher","first-page":"2494","DOI":"10.1109\/TIP.2017.2676345","volume":"26","author":"X Xu","year":"2017","unstructured":"Xu X, Shen F, Yang Y, Shen HT, Li X (2017) Learning discriminative binary codes for large-scale cross-modal retrieval. IEEE Trans Image Process 26(5):2494\u20132507","journal-title":"IEEE Trans Image Process"},{"key":"7343_CR50","doi-asserted-by":"publisher","unstructured":"Xu X, Song J, Lu H, Yang Y, Shen F, Zi H (2018) Modal-adversarial semantic learning network for extendable cross-modal retrieval. In: International conference on multimedia retrieval (ICMR), Yokohama, Japan, June 11\u201314, pp 46\u201354. \nhttps:\/\/doi.org\/10.1145\/3206025.3206033","DOI":"10.1145\/3206025.3206033"},{"key":"7343_CR51","doi-asserted-by":"crossref","unstructured":"Yao T, Mei T, Ngo CW (2015) Learning query and image similarities with ranking canonical correlation analysis. In: IEEE International conference on computer vision (ICCV), pp 28\u201336","DOI":"10.1109\/ICCV.2015.12"},{"issue":"6","key":"7343_CR52","doi-asserted-by":"publisher","first-page":"965","DOI":"10.1109\/TCSVT.2013.2276704","volume":"24","author":"X Zhai","year":"2014","unstructured":"Zhai X, Peng Y, Xiao J (2014) Learning cross-media joint representation with sparse and semisupervised regularization. IEEE Trans Circuits Syst Video Technol 24 (6):965\u2013978","journal-title":"IEEE Trans Circuits Syst Video Technol"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-019-7343-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11042-019-7343-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-019-7343-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,6,6]],"date-time":"2020-06-06T17:16:31Z","timestamp":1591463791000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11042-019-7343-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,2,21]]},"references-count":52,"journal-issue":{"issue":"21-22","published-print":{"date-parts":[[2020,6]]}},"alternative-id":["7343"],"URL":"https:\/\/doi.org\/10.1007\/s11042-019-7343-8","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"value":"1380-7501","type":"print"},{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,2,21]]},"assertion":[{"value":"14 August 2018","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 January 2019","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 February 2019","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 February 2019","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}