{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,8]],"date-time":"2025-09-08T06:34:53Z","timestamp":1757313293136,"version":"3.37.3"},"reference-count":48,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2018,12,19]],"date-time":"2018-12-19T00:00:00Z","timestamp":1545177600000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100001809","name":"Chinese National Natural Science Foundation","doi-asserted-by":"crossref","award":["61532018"],"award-info":[{"award-number":["61532018"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"Chinese National Natural Science Foundation","doi-asserted-by":"crossref","award":["61471049"],"award-info":[{"award-number":["61471049"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2019,6]]},"DOI":"10.1007\/s11042-018-7068-0","type":"journal-article","created":{"date-parts":[[2018,12,19]],"date-time":"2018-12-19T12:56:46Z","timestamp":1545224206000},"page":"16615-16631","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":11,"title":["Two-stage deep learning for supervised cross-modal retrieval"],"prefix":"10.1007","volume":"78","author":[{"given":"Jie","family":"Shao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhicheng","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fei","family":"Su","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2018,12,19]]},"reference":[{"key":"7068_CR1","unstructured":"Andrew G, Arora R, Bilmes J, Livescu K (2013) Deep canonical correlation analysis. In: Proceedings of the 30th international conference on machine learning, pp 1247\u20131255"},{"key":"7068_CR2","doi-asserted-by":"publisher","first-page":"322","DOI":"10.1016\/j.neucom.2015.12.039","volume":"182","author":"J Cai","year":"2016","unstructured":"Cai J, Tang Y, Wang J (2016) Kernel canonical correlation analysis via gradient descent. Neurocomputing 182:322\u2013331","journal-title":"Neurocomputing"},{"key":"7068_CR3","doi-asserted-by":"crossref","unstructured":"Chua TS, Tang J, Hong R, Li H, Luo Z, Zheng Y (2009) Nus-wide: a real-world web image database from national university of singapore. In: Proceedings of the ACM international conference on image and video retrieval. ACM, p 48","DOI":"10.1145\/1646396.1646452"},{"issue":"3","key":"7068_CR4","doi-asserted-by":"publisher","first-page":"521","DOI":"10.1109\/TPAMI.2013.142","volume":"36","author":"PJ Costa","year":"2013","unstructured":"Costa PJ, Coviello E, Doyle G, Rasiwasia N, Lanckriet GR, Levy R, Vasconcelos N (2013) On the role of correlation and abstraction in cross-modal multimedia retrieval. IEEE Trans Pattern Anal Mach Intell 36(3):521\u201335","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"issue":"3","key":"7068_CR5","doi-asserted-by":"publisher","first-page":"521","DOI":"10.1109\/TPAMI.2013.142","volume":"36","author":"J Costa Pereira","year":"2014","unstructured":"Costa Pereira J, Coviello E, Doyle G, Rasiwasia N, Lanckriet GR, Levy R, Vasconcelos N (2014) On the role of correlation and abstraction in cross-modal multimedia retrieval. IEEE Trans Pattern Anal Mach Intell 36(3):521\u2013535","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"7068_CR6","doi-asserted-by":"crossref","unstructured":"Feng F, Wang X, Li R (2014) Cross-modal retrieval with correspondence autoencoder. In: Proceedings of the ACM international conference on multimedia. ACM, pp 7\u201316","DOI":"10.1145\/2647868.2654902"},{"key":"7068_CR7","doi-asserted-by":"publisher","first-page":"50","DOI":"10.1016\/j.neucom.2014.12.020","volume":"154","author":"F Feng","year":"2015","unstructured":"Feng F, Li R, Wang X (2015) Deep correspondence restricted boltzmann machine for cross-modal retrieval. Neurocomputing 154:50\u201360","journal-title":"Neurocomputing"},{"key":"7068_CR8","unstructured":"Frome A, Corrado GS, Shlens J, Bengio S, Dean J, Mikolov T et al (2013) Devise: a deep visual-semantic embedding model. In: Advances in neural information processing systems, pp 2121\u20132129"},{"key":"7068_CR9","doi-asserted-by":"publisher","first-page":"83","DOI":"10.1016\/j.sigpro.2014.08.034","volume":"112","author":"Z Gao","year":"2015","unstructured":"Gao Z, Zhang H, Xu G, Xue Y, Hauptmann AG (2015) Multi-view discriminative and structured dictionary learning with group sparsity for human action recognition. Signal Process 112:83\u201397","journal-title":"Signal Process"},{"key":"7068_CR10","doi-asserted-by":"crossref","unstructured":"Gao Z, Wang D, He X, Zhang H (2018) Group-pair convolutional neural networks for multi-view based 3d object retrieval","DOI":"10.1609\/aaai.v32i1.11899"},{"issue":"2","key":"7068_CR11","doi-asserted-by":"publisher","first-page":"210","DOI":"10.1007\/s11263-013-0658-4","volume":"106","author":"Y Gong","year":"2014","unstructured":"Gong Y, Ke Q, Isard M, Lazebnik S (2014) A multi-view embedding space for modeling internet images, tags, and their semantics. Int J Comput Vis 106 (2):210\u2013233","journal-title":"Int J Comput Vis"},{"issue":"8","key":"7068_CR12","doi-asserted-by":"publisher","first-page":"1371","DOI":"10.1109\/TPAMI.2007.70791","volume":"30","author":"D Grangier","year":"2008","unstructured":"Grangier D, Bengio S (2008) A discriminative kernel-based approach to rank images from text queries. IEEE Trans Pattern Anal Mach Intell 30(8):1371\u20131384","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"7068_CR13","doi-asserted-by":"crossref","unstructured":"Hadsell R, Chopra S, Lecun Y (2006) Dimensionality reduction by learning an invariant mapping. In: 2006 IEEE computer society conference on computer vision and pattern recognition, pp 1735\u20131742","DOI":"10.1109\/CVPR.2006.100"},{"issue":"12","key":"7068_CR14","doi-asserted-by":"publisher","first-page":"2639","DOI":"10.1162\/0899766042321814","volume":"16","author":"DR Hardoon","year":"2004","unstructured":"Hardoon DR, Szedmak S, Shawetaylor J (2004) Canonical correlation analysis: an overview with application to learning methods. Neural Comput 16(12):2639","journal-title":"Neural Comput"},{"key":"7068_CR15","unstructured":"Hinton GE, Salakhutdinov R (2009) Replicated softmax: an undirected topic model. In: Advances in neural information processing systems, pp 1607\u20131614"},{"key":"7068_CR16","doi-asserted-by":"crossref","unstructured":"Huang X, Peng Y (2017) Cross-modal deep metric learning with multi-task regularization. arXiv: http:\/\/arXiv.org\/abs\/1703.07026","DOI":"10.1109\/ICME.2017.8019340"},{"key":"7068_CR17","doi-asserted-by":"crossref","unstructured":"Jia Y, Shelhamer E, Donahue J, Karayev S, Long J, Girshick R, Guadarrama S, Darrell T (2014) Caffe: Convolutional architecture for fast feature embedding. arXiv: http:\/\/arXiv.org\/abs\/1408.5093","DOI":"10.1145\/2647868.2654889"},{"key":"7068_CR18","unstructured":"Kang C, Liao S, He Y, Wang J, Xiang S, Pan C (2014) Cross-modal similarity learning: a low rank bilinear formulation. arXiv: http:\/\/arXiv.org\/abs\/1411.4738"},{"key":"7068_CR19","doi-asserted-by":"crossref","unstructured":"Li D, Dimitrova N, Li M, Sethi IK (2003) Multimedia content processing through cross-modal association. In: Proceedings of the eleventh ACM international conference on multimedia. ACM, pp 604\u2013611","DOI":"10.1145\/957013.957143"},{"key":"7068_CR20","unstructured":"Ngiam J, Khosla A, Kim M, Nam J, Lee H, Ng AY (2011) Multimodal deep learning. In: Proceedings of the 28th international conference on machine learning (ICML-11), pp 689\u2013696"},{"issue":"1","key":"7068_CR21","doi-asserted-by":"publisher","first-page":"75","DOI":"10.1007\/s00530-014-0394-9","volume":"22","author":"W Nie","year":"2016","unstructured":"Nie W, Liu A, Su Y (2016) Cross-domain semantic transfer from large-scale social media. Multimed Syst 22(1):75\u201385","journal-title":"Multimed Syst"},{"key":"7068_CR22","unstructured":"Peng Y, Huang X, Qi J (2016) Cross-media shared representation by hierarchical learning with multiple deep networks. In: International joint conference on artificial intelligence (IJCAI), pp 3846\u20133853"},{"key":"7068_CR23","doi-asserted-by":"crossref","unstructured":"Rasiwasia N, Costa Pereira J, Coviello E, Doyle G, Lanckriet GR, Levy R, Vasconcelos N (2010) A new approach to cross-modal multimedia retrieval. In: Proceedings of the international conference on multimedia. ACM, pp 251\u2013260","DOI":"10.1145\/1873951.1873987"},{"key":"7068_CR24","doi-asserted-by":"crossref","unstructured":"Rosipal R, Kr\u00e4mer N (2006) Overview and recent advances in partial least squares. In: Subspace, latent structure and feature selection. Springer, pp 34\u201351","DOI":"10.1007\/11752790_2"},{"key":"7068_CR25","doi-asserted-by":"crossref","unstructured":"Shao J, Zhao Z, Su F, Yue T (2015) 3view deep canonical correlation analysis for cross-modal retrieval. In: Visual communications and image processing (VCIP), 2015. IEEE, pp 1\u20134","DOI":"10.1109\/VCIP.2015.7457870"},{"key":"7068_CR26","doi-asserted-by":"publisher","first-page":"618","DOI":"10.1016\/j.neucom.2016.06.047","volume":"214","author":"J Shao","year":"2016","unstructured":"Shao J, Wang L, Zhao Z, Cai A et al (2016) Deep canonical correlation analysis with progressive and hypergraph learning for cross-modal retrieval. Neurocomputing 214:618\u2013628","journal-title":"Neurocomputing"},{"key":"7068_CR27","doi-asserted-by":"crossref","unstructured":"Sharma A, Kumar A, Daume H III, Jacobs DW (2012) Generalized multiview analysis: a discriminative latent space. In: 2012 IEEE conference on computer vision and pattern recognition (CVPR). IEEE, pp 2160\u20132167","DOI":"10.1109\/CVPR.2012.6247923"},{"key":"7068_CR28","volume-title":"Parallel distributed processing: explorations in the microstructure of cognition, vol. 1. chapter information processing in dynamical systems: foundations of harmony theory","author":"P Smolensky","year":"1986","unstructured":"Smolensky P (1986) Parallel distributed processing: explorations in the microstructure of cognition, vol. 1. chapter information processing in dynamical systems: foundations of harmony theory. MIT Press, Cambridge. 15, 18"},{"key":"7068_CR29","unstructured":"Socher R, Ganjoo M, Manning CD, Ng A (2013) Zero-shot learning through cross-modal transfer. In: Advances in neural information processing systems, pp 935\u2013943"},{"key":"7068_CR30","doi-asserted-by":"crossref","unstructured":"Song J, Yang Y, Yang Y, Huang Z, Shen HT (2013) Inter-media hashing for large-scale retrieval from heterogeneous data sources. In: Proceedings of the 2013 ACM SIGMOD international conference on management of data. ACM, pp 785\u2013796","DOI":"10.1145\/2463676.2465274"},{"issue":"11","key":"7068_CR31","doi-asserted-by":"publisher","first-page":"4999","DOI":"10.1109\/TIP.2016.2601260","volume":"25","author":"J Song","year":"2016","unstructured":"Song J, Gao L, Nie F, Shen HT, Yan Y, Sebe N (2016) Optimized graph learning using partial tags and multiple features for image and video annotation. IEEE Trans Image Process 25(11):4999\u20135011","journal-title":"IEEE Trans Image Process"},{"issue":"99","key":"7068_CR32","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/TNNLS.2018.2851077","volume":"PP","author":"J Song","year":"2018","unstructured":"Song J, Guo Y, Gao L, Li X, Hanjalic A, Shen HT (2018) From deterministic to generative: multimodal stochastic rnns for video captioning. IEEE Trans Neural Netw Learn Syst PP(99):1\u201312. https:\/\/doi.org\/10.1109\/TNNLS.2018.2851077","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"issue":"7","key":"7068_CR33","doi-asserted-by":"publisher","first-page":"3210","DOI":"10.1109\/TIP.2018.2814344","volume":"27","author":"J Song","year":"2018","unstructured":"Song J, Zhang H, Li X, Gao L, Wang M, Hong R (2018) Self-supervised video hashing with hierarchical binary auto-encoder. IEEE Trans Image Process 27 (7):3210\u20133221","journal-title":"IEEE Trans Image Process"},{"key":"7068_CR34","unstructured":"Srivastava N, Salakhutdinov R (2012) Learning representations for multimodal data with deep belief nets. In: International conference on machine learning workshop"},{"issue":"7-8","key":"7068_CR35","doi-asserted-by":"publisher","first-page":"2031","DOI":"10.1007\/s00521-013-1362-6","volume":"23","author":"S Sun","year":"2013","unstructured":"Sun S (2013) A survey of multi-view machine learning. Neural Comput & Appl 23(7-8):2031\u20132038","journal-title":"Neural Comput & Appl"},{"issue":"16","key":"7068_CR36","doi-asserted-by":"publisher","first-page":"2980","DOI":"10.1016\/j.neucom.2010.07.007","volume":"73","author":"S Sun","year":"2010","unstructured":"Sun S, Hardoon DR (2010) Active learning with extremely sparse labeled examples. Neurocomputing 73(16):2980\u20132988","journal-title":"Neurocomputing"},{"key":"7068_CR37","unstructured":"Sun Y, Chen Y, Wang X, Tang X (2014) Deep learning face representation by joint identification-verification. In: Advances in neural information processing systems, pp 1988\u20131996"},{"issue":"6","key":"7068_CR38","doi-asserted-by":"publisher","first-page":"1247","DOI":"10.1162\/089976600300015349","volume":"12","author":"JB Tenenbaum","year":"2000","unstructured":"Tenenbaum JB, Freeman WT (2000) Separating style and content with bilinear models. Neural computation 12(6):1247\u20131283","journal-title":"Neural computation"},{"issue":"4","key":"7068_CR39","doi-asserted-by":"publisher","first-page":"510","DOI":"10.1109\/LSP.2016.2611485","volume":"24","author":"X Wang","year":"2017","unstructured":"Wang X, Gao L, Song J, Shen H (2017) Beyond frame-level cnn: saliency-aware 3-d cnn with lstm for video action recognition. IEEE Signal Process Lett 24(4):510\u2013514","journal-title":"IEEE Signal Process Lett"},{"issue":"3","key":"7068_CR40","doi-asserted-by":"publisher","first-page":"634","DOI":"10.1109\/TMM.2017.2749159","volume":"20","author":"X Wang","year":"2018","unstructured":"Wang X, Gao L, Wang P, Sun X, Liu X (2018) Two-stream 3-d convnet fusion for action recognition in videos with arbitrary size and length. IEEE Trans Multimed 20(3):634\u2013644","journal-title":"IEEE Trans Multimed"},{"issue":"4","key":"7068_CR41","first-page":"57","volume":"7","author":"Y Wei","year":"2016","unstructured":"Wei Y, Zhao Y, Zhu Z, Wei S, Xiao Y, Feng J, Yan S (2016) Modality-dependent cross-media retrieval. ACM Trans Intell Syst Technol (TIST) 7 (4):57","journal-title":"ACM Trans Intell Syst Technol (TIST)"},{"key":"7068_CR42","unstructured":"Welling M, Rosen-Zvi M, Hinton GE (2004) Exponential family harmoniums with an application to information retrieval. In: Advances in neural information processing systems, pp 1481\u20131488"},{"key":"7068_CR43","doi-asserted-by":"crossref","unstructured":"Wen Y, Zhang K, Li Z, Qiao Y (2016) A discriminative feature learning approach for deep face recognition. In: European conference on computer vision. Springer, pp 499\u2013515","DOI":"10.1007\/978-3-319-46478-7_31"},{"issue":"1","key":"7068_CR44","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1007\/s10994-010-5198-3","volume":"81","author":"J Weston","year":"2010","unstructured":"Weston J, Bengio S, Usunier N (2010) Large scale image annotation: learning to rank with joint word-image embeddings. Mach Learn 81(1):21\u201335","journal-title":"Mach Learn"},{"key":"7068_CR45","doi-asserted-by":"crossref","unstructured":"Wu F, Lu X, Zhang Z, Yan S, Rui Y, Zhuang Y (2013) Cross-media semantic representation via bi-directional learning to rank. In: Proceedings of the 21st ACM international conference on multimedia. ACM, pp 877\u2013886","DOI":"10.1145\/2502081.2502097"},{"issue":"6","key":"7068_CR46","doi-asserted-by":"publisher","first-page":"965","DOI":"10.1109\/TCSVT.2013.2276704","volume":"24","author":"X Zhai","year":"2014","unstructured":"Zhai X, Peng Y, Xiao J (2014) Learning cross-media joint representation with sparse and semisupervised regularization. IEEE Trans Circuits Syst Video Technol 24 (6):965\u2013978","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"issue":"9","key":"7068_CR47","doi-asserted-by":"publisher","first-page":"2033","DOI":"10.1109\/TMM.2017.2703636","volume":"19","author":"X Zhu","year":"2017","unstructured":"Zhu X, Li X, Zhang S, Xu Z, Yu L, Wang C (2017) Graph pca hashing for similarity search. IEEE Trans Multimed 19(9):2033\u20132044","journal-title":"IEEE Trans Multimed"},{"key":"7068_CR48","doi-asserted-by":"publisher","first-page":"263","DOI":"10.1016\/j.neucom.2016.01.053","volume":"191","author":"C Zu","year":"2016","unstructured":"Zu C, Zhang D (2016) Canonical sparse cross-view correlation analysis. Neurocomputing 191:263\u2013272","journal-title":"Neurocomputing"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-018-7068-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11042-018-7068-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-018-7068-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,9,8]],"date-time":"2022-09-08T19:54:02Z","timestamp":1662666842000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11042-018-7068-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,12,19]]},"references-count":48,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2019,6]]}},"alternative-id":["7068"],"URL":"https:\/\/doi.org\/10.1007\/s11042-018-7068-0","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"type":"print","value":"1380-7501"},{"type":"electronic","value":"1573-7721"}],"subject":[],"published":{"date-parts":[[2018,12,19]]},"assertion":[{"value":"25 April 2018","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 November 2018","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 December 2018","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 December 2018","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}