{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T09:51:44Z","timestamp":1779270704632,"version":"3.51.4"},"reference-count":53,"publisher":"Springer Science and Business Media LLC","issue":"21","license":[{"start":{"date-parts":[[2018,9,6]],"date-time":"2018-09-06T00:00:00Z","timestamp":1536192000000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"name":"National Social Science Foundation of China","award":["15BGL048"],"award-info":[{"award-number":["15BGL048"]}]},{"name":"Hubei Province Science and Technology Support Project","award":["2015BAA072"],"award-info":[{"award-number":["2015BAA072"]}]},{"name":"Hubei Provincial Natural Science Foundation of China","award":["2017CFA012"],"award-info":[{"award-number":["2017CFA012"]}]},{"name":"The Fundamental Research Funds for the Central Universities","award":["2017II39GX"],"award-info":[{"award-number":["2017II39GX"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2019,11]]},"DOI":"10.1007\/s11042-018-6556-6","type":"journal-article","created":{"date-parts":[[2018,9,6]],"date-time":"2018-09-06T01:26:27Z","timestamp":1536197187000},"page":"30769-30791","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Enhancing multimodal deep representation learning by fixed model reuse"],"prefix":"10.1007","volume":"78","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6346-0707","authenticated-orcid":false,"given":"Zhongwei","family":"Xie","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lin","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xian","family":"Zhong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yang","family":"He","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Luo","family":"Zhong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2018,9,6]]},"reference":[{"key":"6556_CR1","unstructured":"Ah-Pine J, Csurka G (2011) Semantic combination of textual and visual information in multimedia retrieval. In: ACM International conference on multimedia retrieval, p 44"},{"key":"6556_CR2","unstructured":"Andrew G, Arora R, Bilmes J, Livescu K (2013) Deep canonical correlation analysis. In: International conference on international conference on machine learning, pp III\u20131247"},{"key":"6556_CR3","doi-asserted-by":"crossref","unstructured":"Baird HS (1993) Document image defect models and their uses. In: 2Nd international conference document analysis and recognition, ICDAR \u201993, october 20-22, 1993, tsukuba city, japan, pp 62\u201367","DOI":"10.1109\/ICDAR.1993.395781"},{"issue":"2","key":"6556_CR4","doi-asserted-by":"publisher","first-page":"157","DOI":"10.1109\/72.279181","volume":"5","author":"Y Bengio","year":"2002","unstructured":"Bengio Y, Simard P, Frasconi P (2002) Learning long-term dependencies with gradient descent is difficult. IEEE Trans Neural Netw 5(2):157\u2013166","journal-title":"IEEE Trans Neural Netw"},{"issue":"8","key":"6556_CR5","doi-asserted-by":"publisher","first-page":"1798","DOI":"10.1109\/TPAMI.2013.50","volume":"35","author":"Y Bengio","year":"2013","unstructured":"Bengio Y, Courville A, Vincent P (2013) Representation learning: a review and new perspectives. IEEE Trans Pattern Anal Mach Intell 35(8):1798\u20131828","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"6556_CR6","doi-asserted-by":"crossref","unstructured":"Breuel TM (2008) The ocropus open source OCR system. In: Document recognition and retrieval XV, part of the IS&T-SPIE Electronic Imaging Symposium, San Jose, CA, USA, January 29-31, 2008. Proceedings, p 68150F","DOI":"10.1117\/12.783598"},{"key":"6556_CR7","doi-asserted-by":"crossref","unstructured":"Chrupala G, Gelderloos L, Alishahi A (2017) Representations of language in a model of visually grounded speech signal. In: Proceedings of the 55th Annual Meeting of the Association for Computational Linguistics, ACL 2017, Vancouver, Canada, July 30 - August 4, Volume 1: Long Papers, pp 613\u2013 622","DOI":"10.18653\/v1\/P17-1057"},{"key":"6556_CR8","unstructured":"Firmani D, Merialdo P, Nieddu E, Scardapane S (2017) In codice ratio: OCR of handwritten latin documents using deep convolutional networks. In: International workshop on artificial intelligence for cultural heritage, pp 9\u201316"},{"key":"6556_CR9","unstructured":"Frome A, Corrado GS, Shlens J, Bengio S, Dean J, Ranzato M, Mikolov T (2013) Devise: a deep visual-semantic embedding model. In: International conference on neural information processing systems, pp 2121\u20132129"},{"key":"6556_CR10","unstructured":"Graves A, Gomez F (2006) Connectionist temporal classification:labelling unsegmented sequence data with recurrent neural networks. In: International conference on machine learning, pp 369\u2013376"},{"issue":"5","key":"6556_CR11","doi-asserted-by":"publisher","first-page":"855","DOI":"10.1109\/TPAMI.2008.137","volume":"31","author":"A Graves","year":"2009","unstructured":"Graves A, Liwicki M, Fernndez S, Bertolami R, Bunke H, Schmidhuber J (2009) A novel connectionist system for unconstrained handwriting recognition. IEEE Trans Pattern Anal Mach Intell 31(5):855","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"issue":"12","key":"6556_CR12","doi-asserted-by":"publisher","first-page":"2639","DOI":"10.1162\/0899766042321814","volume":"16","author":"DR Hardoon","year":"2014","unstructured":"Hardoon DR, Szedmak S, Shawe-Taylor J (2014) Canonical correlation analysis: an overview with application to learning methods. Neural Comput 16(12):2639\u20132664","journal-title":"Neural Comput"},{"issue":"9","key":"6556_CR13","doi-asserted-by":"publisher","first-page":"1904","DOI":"10.1109\/TPAMI.2015.2389824","volume":"37","author":"K He","year":"2015","unstructured":"He K, Zhang X, Ren S, Sun J (2015) Spatial pyramid pooling in deep convolutional networks for visual recognition. IEEE Trans Pattern Anal Mach Intell 37 (9):1904\u20131916","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"issue":"8","key":"6556_CR14","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter S, Schmidhuber J (1997) Long short-term memory. Neural Comput 9(8):1735\u20131780","journal-title":"Neural Comput"},{"key":"6556_CR15","unstructured":"Kiros R, Salakhutdinov R, Zemel RS (2014) Unifying visual-semantic embeddings with multimodal neural language models Computer Science"},{"key":"6556_CR16","doi-asserted-by":"crossref","unstructured":"Klein B, Lev G, Sadeh G, Wolf L (2015) Associating neural word embeddings with deep image representations using fisher vectors. In: IEEE Conference on computer vision and pattern recognition, pp 4437\u20134446","DOI":"10.1109\/CVPR.2015.7299073"},{"key":"6556_CR17","volume-title":"NAACL HLT 2016, The 2016 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","year":"2016","unstructured":"Knight K, Nenkova A, Rambow O (eds) (2016) NAACL HLT 2016, The 2016 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. The Association for Computational Linguistics, San Diego"},{"key":"6556_CR18","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2012) Imagenet classification with deep convolutional neural networks. In: International conference on neural information processing systems, pp 1097\u20131105"},{"key":"6556_CR19","unstructured":"Le QV, Mikolov T (2014) Distributed representations of sentences and documents. In: Proceedings of the 31th International Conference on Machine Learning, ICML 2014, Beijing, China, 21-26 June 2014, pp 1188\u20131196"},{"key":"6556_CR20","doi-asserted-by":"crossref","unstructured":"Li D, Dimitrova N, Li M, Sethi IK (2003) Multimedia content processing through cross-modal association. In: Proceedings of the Eleventh ACM International Conference on Multimedia, Berkeley, CA, USA, November 2-8, 2003, pp 604\u2013611","DOI":"10.1145\/957013.957143"},{"issue":"6","key":"6556_CR21","doi-asserted-by":"publisher","first-page":"1370","DOI":"10.1109\/TPAMI.2012.172","volume":"35","author":"N Li","year":"2013","unstructured":"Li N, Tsang IW, Zhou ZH (2013) Efficient optimization of performance measures by classifier adaptation. IEEE Trans Pattern Anal Mach Intell 35(6):1370\u20131382","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"6556_CR22","doi-asserted-by":"crossref","unstructured":"Liu Y, Zhao WL, Ngo CW, Xu CS, Lu HQ (2010) Coherent bag-of audio words model for efficient large-scale video copy detection. In: Acm international conference on image & video retrieval, pp 89\u201396","DOI":"10.1145\/1816041.1816057"},{"key":"6556_CR23","unstructured":"Long M, Cao Y, Wang J, Jordan MI (2015) Learning transferable features with deep adaptation networks. In: International conference on international conference on machine learning, pp 97\u2013105"},{"key":"6556_CR24","unstructured":"Mihalcea R, Tarau P (2004) Textrank: Bringing order into texts. Emnlp, pp 404\u2013411"},{"issue":"19","key":"6556_CR25","doi-asserted-by":"publisher","first-page":"2016","DOI":"10.19026\/rjaset.8.1193","volume":"8","author":"S Naz","year":"2014","unstructured":"Naz S, Umar AI, Shirazi SH, Ajmal MM (2014) Salahuddin: The optical character recognition for cursive script using hmm: A review. Res J Appl Sci Eng Technol 8(19):2016\u20132025","journal-title":"Res J Appl Sci Eng Technol"},{"key":"6556_CR26","unstructured":"Ngiam J, Khosla A, Kim M, Nam J, Lee H, Ng AY (2009) Multimodal deep learning. In: International conference on machine learning, ICML 2011, bellevue, washington, usa, june 28 - july, pp 689\u2013696"},{"issue":"10","key":"6556_CR27","doi-asserted-by":"publisher","first-page":"1345","DOI":"10.1109\/TKDE.2009.191","volume":"22","author":"SJ Pan","year":"2010","unstructured":"Pan SJ, Yang Q (2010) A survey on transfer learning. IEEE Trans Knowl Data Eng 22(10):1345\u20131359","journal-title":"IEEE Trans Knowl Data Eng"},{"key":"6556_CR28","doi-asserted-by":"crossref","unstructured":"Pan Y, Mei T, Yao T, Li H, Rui Y (2016) Jointly modeling embedding and translation to bridge video and language. In: Computer vision and pattern recognition, pp 4594\u20134602","DOI":"10.1109\/CVPR.2016.497"},{"issue":"99","key":"6556_CR29","first-page":"1","volume":"PP","author":"Y Peng","year":"2017","unstructured":"Peng Y, Huang X, Zhao Y (2017) An overview of cross-media retrieval: concepts, methodologies, benchmarks and challenges. IEEE Trans Circuits Syst Video Technol PP(99):1\u20131","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"key":"6556_CR30","doi-asserted-by":"crossref","unstructured":"Philip B, Samuel RDS (2009) A novel bilingual ocr system based on column-stochastic features and svm classifier for the specially enabled. In: Second international conference on emerging trends in engineering & technology, pp 252\u2013257","DOI":"10.1109\/ICETET.2009.14"},{"key":"6556_CR31","doi-asserted-by":"crossref","unstructured":"Rasiwasia N, Pereira JC, Coviello E, Doyle G, Lanckriet GRG, Levy R, Vasconcelos N (2010) A new approach to cross-modal multimedia retrieval. In: International conference on multimedia, pp 251\u2013260","DOI":"10.1145\/1873951.1873987"},{"key":"6556_CR32","doi-asserted-by":"crossref","unstructured":"Silberer C, Lapata M (2014) Learning grounded meaning representations with autoencoders. In: Meeting of the association for computational linguistics, pp 721\u2013732","DOI":"10.3115\/v1\/P14-1068"},{"key":"6556_CR33","doi-asserted-by":"crossref","unstructured":"Smith R, Antonova D, Lee DS (2009) Adapting the tesseract open source ocr engine for multilingual ocr. In: International workshop on multilingual ocr, p 1","DOI":"10.1145\/1577802.1577804"},{"key":"6556_CR34","doi-asserted-by":"crossref","first-page":"207","DOI":"10.1162\/tacl_a_00177","volume":"2","author":"R Socher","year":"2014","unstructured":"Socher R, Karpathy A, Le QV, Manning CD, Ng AY (2014) Grounded compositional semantics for finding and describing images with sentences. TACL 2:207\u2013218","journal-title":"TACL"},{"issue":"2","key":"6556_CR35","doi-asserted-by":"publisher","first-page":"101","DOI":"10.1007\/s41019-016-0012-2","volume":"1","author":"R Song","year":"2016","unstructured":"Song R, Umemoto K, Nie J, Xie X, Tanaka K, Rui Y (2016) Uniclip: Leveraging web search for universal clipping of articles on mobile. Data Science and Engineering 1(2):101\u2013113","journal-title":"Data Science and Engineering"},{"key":"6556_CR36","unstructured":"Srivastava N, Salakhutdinov R (2012) Learning representations for multimodal data with deep belief nets. In: International conference on machine learning"},{"issue":"1","key":"6556_CR37","first-page":"1929","volume":"15","author":"N Srivastava","year":"2014","unstructured":"Srivastava N, Hinton G, Krizhevsky A, Sutskever I, Salakhutdinov R (2014) Dropout: a simple way to prevent neural networks from overfitting. J Mach Learn Res 15(1):1929\u20131958","journal-title":"J Mach Learn Res"},{"key":"6556_CR38","doi-asserted-by":"crossref","unstructured":"Szegedy C, Vanhoucke V, Ioffe S, Shlens J, Wojna Z (2016) Rethinking the inception architecture for computer vision. In: 2016 IEEE Conference on computer vision and pattern recognition, CVPR 2016, las vegas, NV, USA, June 27-30, 2016, pp 2818\u20132826","DOI":"10.1109\/CVPR.2016.308"},{"key":"6556_CR39","doi-asserted-by":"crossref","unstructured":"Ul-Hasan A, Breuel TM (2013) Can we build language-independent ocr using lstm networks?. In: International workshop on multilingual ocr, p 9","DOI":"10.1145\/2505377.2505394"},{"key":"6556_CR40","doi-asserted-by":"crossref","unstructured":"Vinyals O, Toshev A, Bengio S, Erhan D (2015) Show and tell: a neural image caption generator. In: IEEE Conference on computer vision and pattern recognition, CVPR 2015, boston, MA, USA, June 7-12, 2015, pp 3156\u20133164","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"6556_CR41","unstructured":"Wang D, Cui P, Ou M, Zhu W (2015) Deep multimodal hashing with orthogonal regularization. In: International conference on artificial intelligence, pp 2291\u20132297"},{"key":"6556_CR42","doi-asserted-by":"crossref","unstructured":"Wang Y, Lin X, Wu L, Zhang W (2015) Effective multi-query expansions: Robust landmark retrieval. In: Proceedings of the 23rd Annual ACM Conference on Multimedia Conference, MM \u201915, Brisbane, Australia, October 26 - 30, 2015, pp 79\u201388","DOI":"10.1145\/2733373.2806233"},{"issue":"11","key":"6556_CR43","doi-asserted-by":"publisher","first-page":"3939","DOI":"10.1109\/TIP.2015.2457339","volume":"24","author":"Y Wang","year":"2015","unstructured":"Wang Y, Lin X, Wu L, Zhang W, Zhang Q, Huang X (2015) Robust subspace clustering for multi-view data by exploiting correlation consensus. IEEE Trans Image Process 24(11):3939\u20133949","journal-title":"IEEE Trans Image Process"},{"issue":"3","key":"6556_CR44","doi-asserted-by":"publisher","first-page":"1393","DOI":"10.1109\/TIP.2017.2655449","volume":"26","author":"Y Wang","year":"2017","unstructured":"Wang Y, Lin X, Wu L, Zhang W (2017) Effective multi-query expansions: Collaborative deep networks for robust landmark retrieval. IEEE Trans Image Processing 26(3):1393\u20131404","journal-title":"IEEE Trans Image Processing"},{"issue":"1","key":"6556_CR45","doi-asserted-by":"publisher","first-page":"57","DOI":"10.1109\/TNNLS.2015.2498149","volume":"28","author":"Y Wang","year":"2017","unstructured":"Wang Y, Zhang W, Wu L, Lin X, Zhao X (2017) Unsupervised metric fusion over multiview data by graph random walk-based cross-view diffusion. IEEE Trans Neural Netw Learn Syst 28(1):57\u201370","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"6556_CR46","unstructured":"Weston J, Bengio S, Usunier N (2011) Wsabie: scaling up to large vocabulary image annotation. In: International joint conference on artificial intelligence, pp 2764\u20132770"},{"key":"6556_CR47","doi-asserted-by":"publisher","first-page":"275","DOI":"10.1016\/j.patcog.2017.08.029","volume":"73","author":"L Wu","year":"2017","unstructured":"Wu L, Wang Y, Gao J, Li X (2017) Deep adaptive feature embedding with local sample distributions for person re-identification. Pattern Recogn 73:275\u2013288","journal-title":"Pattern Recogn"},{"key":"6556_CR48","doi-asserted-by":"crossref","unstructured":"Wu L, Wang Y, Li X, Gao J (2017) What-and-where to match: Deep spatially multiplicative integration networks for person re-identification Pattern Recognition","DOI":"10.1016\/j.patcog.2017.10.004"},{"issue":"99","key":"6556_CR49","first-page":"1","volume":"PP","author":"L Wu","year":"2018","unstructured":"Wu L, Wang Y, Li X, Gao J (2018) Deep attention-based spatially recursive networks for fine-grained visual recognition. IEEE Transactions on Cybernetics PP (99):1\u201312","journal-title":"IEEE Transactions on Cybernetics"},{"key":"6556_CR50","doi-asserted-by":"crossref","unstructured":"Yang Y, Zhan D, Fan Y, Jiang Y, Zhou Z (2017) Deep learning for fixed model reuse. In: Proceedings of the Thirty-First AAAI Conference on Artificial Intelligence, February 4-9, 2017, San Francisco, California, USA., pp 2831\u20132837","DOI":"10.1609\/aaai.v31i1.10855"},{"key":"6556_CR51","unstructured":"Yosinski J, Clune J, Bengio Y, Lipson H (2014) How transferable are features in deep neural networks?. In: Advances in neural information processing systems 27: Annual conference on neural information processing systems 2014, december 8-13 2014, montreal, quebec, canada, pp 3320\u20133328"},{"key":"6556_CR52","doi-asserted-by":"crossref","first-page":"67","DOI":"10.1162\/tacl_a_00166","volume":"2","author":"P Young","year":"2014","unstructured":"Young P, Lai A, Hodosh M, Hockenmaier J (2014) From image descriptions to visual denotations: New similarity metrics for semantic inference over event descriptions. TACL 2:67\u201378","journal-title":"TACL"},{"key":"6556_CR53","volume-title":"Learnware: on the future of machine learning","author":"ZH Zhou","year":"2016","unstructured":"Zhou ZH (2016) Learnware: on the future of machine learning. Springer, New York"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-018-6556-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11042-018-6556-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-018-6556-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,8,31]],"date-time":"2022-08-31T20:28:38Z","timestamp":1661977718000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11042-018-6556-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,9,6]]},"references-count":53,"journal-issue":{"issue":"21","published-print":{"date-parts":[[2019,11]]}},"alternative-id":["6556"],"URL":"https:\/\/doi.org\/10.1007\/s11042-018-6556-6","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"value":"1380-7501","type":"print"},{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2018,9,6]]},"assertion":[{"value":"30 July 2018","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 August 2018","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 August 2018","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 September 2018","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}