{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,4]],"date-time":"2026-04-04T00:38:54Z","timestamp":1775263134935,"version":"3.50.1"},"reference-count":39,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2018,11,5]],"date-time":"2018-11-05T00:00:00Z","timestamp":1541376000000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61603343"],"award-info":[{"award-number":["61603343"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61703372"],"award-info":[{"award-number":["61703372"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"the Science & Technology Innovation Team Project of Henan Province","award":["17IRTSTHN013"],"award-info":[{"award-number":["17IRTSTHN013"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2019,4]]},"DOI":"10.1007\/s10489-018-1337-5","type":"journal-article","created":{"date-parts":[[2018,11,5]],"date-time":"2018-11-05T00:03:34Z","timestamp":1541376214000},"page":"1306-1323","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["Emergent spatio-temporal multimodal learning using a developmental network"],"prefix":"10.1007","volume":"49","author":[{"given":"Dongshu","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1024-4135","authenticated-orcid":false,"given":"Jianbin","family":"Xin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2018,11,5]]},"reference":[{"issue":"3","key":"1337_CR1","doi-asserted-by":"publisher","first-page":"83","DOI":"10.1016\/j.robot.2014.11.005","volume":"71","author":"A Droniou","year":"2015","unstructured":"Droniou A, Ivaldi S, Sigaud O (2015) Deep unsupervised network for multimodal perception, representation and classification. Robot Auton Syst 71(3):83\u201398","journal-title":"Robot Auton Syst"},{"key":"1337_CR2","volume-title":"Continuities and discontinuities in development: chapter 8","author":"BI Bertenthal","year":"1984","unstructured":"Bertenthal BI, Campos JJ, Barrett KC (1984) Continuities and discontinuities in development: chapter 8. Plenum Press, New York"},{"issue":"6669","key":"1337_CR3","doi-asserted-by":"publisher","first-page":"756","DOI":"10.1038\/35784","volume":"391","author":"M Botvinick","year":"1998","unstructured":"Botvinick M, Cohen J (1998) Rubber hands \u2018feel\u2019 touch that eyes see. Nature 391(6669):756","journal-title":"Nature"},{"issue":"1","key":"1337_CR4","doi-asserted-by":"publisher","first-page":"47","DOI":"10.1007\/s10489-007-0074-y","volume":"30","author":"O Brdiczka","year":"2009","unstructured":"Brdiczka O, Maisonnasse J, Reignier P, Crowley JL (2009) Detecting small group activities from multimodal observations. Appl Intell 30(1):47\u201357","journal-title":"Appl Intell"},{"key":"1337_CR5","doi-asserted-by":"publisher","first-page":"30","DOI":"10.1016\/j.cortex.2017.10.003","volume":"99","author":"CA Bareham","year":"2018","unstructured":"Bareham CA, Georgieva SD, Kamke MR, Lloyd D, Bekinschtein TA, Mattingley JB (2018) Role of the right inferior parietal cortex in auditory selective attention: an rTMS study. Cortex 99:30\u201338","journal-title":"Cortex"},{"issue":"3","key":"1337_CR6","first-page":"625","volume":"11","author":"D Erhan","year":"2010","unstructured":"Erhan D, Bengio Y, Courville A, Manzagol P-A, Vincent P (2010) Why does unsupervised pre-training help deep learning? J Mach Learn Res 11(3):625\u2013660","journal-title":"J Mach Learn Res"},{"key":"1337_CR7","unstructured":"Fredenslund K Computational complexity of neural networks. https:\/\/kasperfred.com\/posts\/computational-complexity-of-neural-networks"},{"key":"1337_CR8","doi-asserted-by":"publisher","first-page":"142","DOI":"10.1007\/s10489-017-0968-2","volume":"48","author":"X Han","year":"2018","unstructured":"Han X, Dai Q (2018) Batch-normalized mlpconv-wise supervised pre-training network in network. Appl Intell 48:142\u2013155","journal-title":"Appl Intell"},{"key":"1337_CR9","doi-asserted-by":"crossref","unstructured":"Hennecke ME, Prasad KV, Stork DG (1994) Using deformable templates to infer visual speech dynamics. In: Proceedings of the 1994 28th Asilomar conference on signals, systems, and computers, pp 578-582. Pacific Grove, CA, USA","DOI":"10.1109\/ACSSC.1994.471518"},{"key":"1337_CR10","doi-asserted-by":"crossref","unstructured":"Hennecke ME, Prasad KV, Stork DG (1995) Automatic speech recognition system using acoustic and visual signals. In: Proceedings of the 1995 29th Asilomar conference on signals, systems, and computers, pp 1214-1218. Pacific Grove, CA, USA","DOI":"10.1109\/ACSSC.1995.540892"},{"key":"1337_CR11","first-page":"509","volume":"13","author":"H Hirsch","year":"1971","unstructured":"Hirsch H, Spinelli D (1971) Modification of the distribution of receptive field oroentation in cats by selectively visual exposure during development. Exp Brain Res 13:509\u2013527","journal-title":"Exp Brain Res"},{"key":"1337_CR12","doi-asserted-by":"publisher","first-page":"144","DOI":"10.1016\/j.neucom.2016.10.086","volume":"253","author":"F Huang","year":"2017","unstructured":"Huang F, Zhang S, Zhang J, Yu G (2017) Multimodal learning for topic sentiment analysis in microblogging. Neurocomputing 253:144\u2013153","journal-title":"Neurocomputing"},{"key":"1337_CR13","doi-asserted-by":"crossref","unstructured":"Huang J, Kingsbury B (2013) Audio-visual deep learning for noise robust speech recognition. In: Proceedings of the IEEE international conference on acoustics, speech, and signal processing, pp 7596-7599. Vancouver, Canada","DOI":"10.1109\/ICASSP.2013.6639140"},{"issue":"11","key":"1337_CR14","doi-asserted-by":"publisher","first-page":"1277","DOI":"10.1109\/34.888712","volume":"22","author":"W Hwang","year":"2000","unstructured":"Hwang W, Weng J (2000) Hierarchical discriminant regression. IEEE Trans Pattern Anal Mach Intell 22 (11):1277\u20131293","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"1337_CR15","doi-asserted-by":"publisher","first-page":"69","DOI":"10.1016\/j.sigpro.2017.07.006","volume":"142","author":"I Ariav","year":"2018","unstructured":"Ariav I, Dov D, Cohen I (2018) A deep architecture for audio-visual voice activity detection in the presence of transients. Signal Process 142:69\u201374","journal-title":"Signal Process"},{"issue":"4","key":"1337_CR16","doi-asserted-by":"publisher","first-page":"722","DOI":"10.1007\/s10489-014-0629-7","volume":"42","author":"K Noda","year":"2015","unstructured":"Noda K, Yamaguchi Y, Nakadai K, Okuno HG, Ogata T (2015) Audio-visual speech recognition using deep learning. Appl Intell 42(4):722\u2013737","journal-title":"Appl Intell"},{"key":"1337_CR17","doi-asserted-by":"crossref","unstructured":"Gurban M, Thiran JP, Drugman T, Dutoit T (2008) Dynamic modality weighting for multi-stream hmms in audio-visual speech recognition. In: Proceedings of the 10th international conference on multimodal interfaces, pp 237-240. Chania, Greece","DOI":"10.1145\/1452392.1452442"},{"key":"1337_CR18","doi-asserted-by":"crossref","unstructured":"Mangin O, Oudeyer PY (2013) Learning semantic components from subsymbolic multimodal perception. In: Proceedings of the third joint international conference on development and learning and epigenetic robotics (ICDL), pp 1\u20137","DOI":"10.1109\/DevLrn.2013.6652563"},{"issue":"3","key":"1337_CR19","doi-asserted-by":"publisher","first-page":"245","DOI":"10.1007\/BF00872091","volume":"4","author":"E McDermott","year":"1994","unstructured":"McDermott E, Katagiri S (1994) Prototype-based minimum error training for speech recognition. Appl Intell 4(3):245\u2013256","journal-title":"Appl Intell"},{"key":"1337_CR20","doi-asserted-by":"publisher","first-page":"746","DOI":"10.1038\/264746a0","volume":"264","author":"H McGurk","year":"1976","unstructured":"McGurk H, MacDonald J (1976) Hearing lips and seeing voices. Nature 264:746\u2013748","journal-title":"Nature"},{"key":"1337_CR21","unstructured":"Mercer N (2000) Words and minds: how we use language to think together. Routledge, London, UK"},{"key":"1337_CR22","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.cognition.2017.01.016","volume":"164","author":"NA Smith","year":"2017","unstructured":"Smith NA, Folland NA, Martinez DM, Trainor LJ (2017) Multisensory object perception in infancy: 4-month-olds perceive a mistuned harmonic as a separate auditory and visual object. Cognition 164:1\u20137","journal-title":"Cognition"},{"issue":"3","key":"1337_CR23","doi-asserted-by":"publisher","first-page":"479","DOI":"10.1007\/s10548-013-0333-7","volume":"28","author":"N Altieri","year":"2015","unstructured":"Altieri N, Stevenson RA, Wallace MT, Wenger MJ (2015) Learning to associate auditory and visual stimuli: behavioral and neural mechanisms. Brain Topogr 28(3):479\u2013493","journal-title":"Brain Topogr"},{"key":"1337_CR24","volume-title":"Artificial intelligence a modern approach","author":"SJ Russell","year":"2011","unstructured":"Russell SJ, Norvig P (2011) Artificial intelligence a modern approach, 3rd. Prentice Hall, Inc., New Jersey","edition":"3rd"},{"key":"1337_CR25","doi-asserted-by":"publisher","first-page":"427","DOI":"10.1016\/j.eswa.2017.08.039","volume":"90","author":"S Khan","year":"2017","unstructured":"Khan S, Xu G, Chan R, Yan H (2017) An online spatio-temporal tensor learning model for visual tracking and its applications to facial expression recognition. Expert Syst Appl 90:427\u2013438","journal-title":"Expert Syst Appl"},{"key":"1337_CR26","doi-asserted-by":"publisher","first-page":"31","DOI":"10.1016\/j.knosys.2017.07.014","volume":"134","author":"G Song","year":"2017","unstructured":"Song G, Dai Q (2017) A novel double deep elms ensemble system for time series forecasting. Knowl-Based Syst 134:31\u201349","journal-title":"Knowl-Based Syst"},{"key":"1337_CR27","unstructured":"Stork DG, Wolff G, Levine E (1992) Neural network lipreading system for improved speech recognition. In: Proceedings of the international joint conference on neural networks (IJCNN), pp 286-295, Baltimore, MD, USA"},{"issue":"1","key":"1337_CR28","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1007\/s13748-016-0106-0","volume":"6","author":"D Wang","year":"2017","unstructured":"Wang D, Chen J, Liu L (2017) How internal neurons represent the short context: an emergent perspective. Progress Artif Intell 6(1):67\u201377","journal-title":"Progress Artif Intell"},{"issue":"10","key":"1337_CR29","doi-asserted-by":"publisher","first-page":"4917","DOI":"10.1109\/TNNLS.2017.2762720","volume":"29","author":"D Wang","year":"2018","unstructured":"Wang D, Duan Y, Weng J (2018) Motivated optimal developmental learning for sequential tasks without using rigid time-discounts. IEEE Trans Neural Netw Learn Syst 29(10):4917\u20134931","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"issue":"10","key":"1337_CR30","doi-asserted-by":"publisher","first-page":"47","DOI":"10.14257\/ijsh.2015.9.10.06","volume":"9","author":"D Wang","year":"2015","unstructured":"Wang D, Liu L (2015) Face recognition in complex background: developmental network and synapse maintenance. Int J Smart Home 9(10):47\u201362","journal-title":"Int J Smart Home"},{"issue":"4","key":"1337_CR31","doi-asserted-by":"publisher","first-page":"359","DOI":"10.1007\/s13748-018-0150-z","volume":"7","author":"D Wang","year":"2018","unstructured":"Wang D, Shan H, Tian Y, Liu L (2018) Emergent face orientation recognition with internal neurons of the developmental network. Progress Artif Intell 7(4):359\u2013367","journal-title":"Progress Artif Intell"},{"issue":"2","key":"1337_CR32","doi-asserted-by":"publisher","first-page":"1135","DOI":"10.1007\/s11063-017-9734-z","volume":"48","author":"D Wang","year":"2018","unstructured":"Wang D, Wang J, Liu L (2018) Developmental network: an internal emergent object feature learning. Neural Process Lett 48(2):1135\u20131159","journal-title":"Neural Process Lett"},{"issue":"13","key":"1337_CR33","doi-asserted-by":"publisher","first-page":"2303","DOI":"10.1016\/j.neucom.2006.07.017","volume":"70","author":"J Weng","year":"2007","unstructured":"Weng J (2007) On developmental mental architectures. Neurocomputing 70(13):2303\u20132323","journal-title":"Neurocomputing"},{"issue":"1","key":"1337_CR34","first-page":"13","volume":"1","author":"J Weng","year":"2011","unstructured":"Weng J (2011) Why have we passed neural networks no not abstract well. Nat Intell INNS Mag 1(1):13\u201322","journal-title":"Nat Intell INNS Mag"},{"issue":"1","key":"1337_CR35","doi-asserted-by":"publisher","first-page":"29","DOI":"10.1109\/TAMD.2011.2159113","volume":"4","author":"J Weng","year":"2012","unstructured":"Weng J (2012) Symbolic models and emergent models: a review. IEEE Trans Auton Ment Dev 4(1):29\u201353","journal-title":"IEEE Trans Auton Ment Dev"},{"issue":"2","key":"1337_CR36","doi-asserted-by":"publisher","first-page":"161","DOI":"10.1109\/TAMD.2011.2174636","volume":"4","author":"J Weng","year":"2012","unstructured":"Weng J, Luciw M (2012) Brain-like emergent spatial processing. IEEE Trans Auton Ment Dev 4(2):161\u2013185","journal-title":"IEEE Trans Auton Ment Dev"},{"key":"1337_CR37","doi-asserted-by":"publisher","first-page":"599","DOI":"10.1126\/science.291.5504.599","volume":"291","author":"J Weng","year":"2001","unstructured":"Weng J, McClelland J, Pentland A, Sporns O, Stockman I, Sur M, Thelen E (2001) Autonomous mental development by robots and animals. Science 291:599\u2013600","journal-title":"Science"},{"issue":"10","key":"1337_CR38","doi-asserted-by":"publisher","first-page":"3191","DOI":"10.1016\/j.patcog.2015.04.012","volume":"48","author":"W Zhang","year":"2015","unstructured":"Zhang W, Zhang Y, Ma L, Guan J, Gong S (2015) Multimodal learning for facial expression recognition. Pattern Recogn 48(10):3191\u20133202","journal-title":"Pattern Recogn"},{"issue":"3","key":"1337_CR39","doi-asserted-by":"publisher","first-page":"149","DOI":"10.1109\/TAMD.2010.2051437","volume":"2","author":"Y Zhang","year":"2010","unstructured":"Zhang Y, Weng J (2010) Spatio-temporal developmental learning. IEEE Trans Auton Ment Dev 2(3):149\u2013166","journal-title":"IEEE Trans Auton Ment Dev"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-018-1337-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10489-018-1337-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-018-1337-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,3]],"date-time":"2026-04-03T23:27:34Z","timestamp":1775258854000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10489-018-1337-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,11,5]]},"references-count":39,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2019,4]]}},"alternative-id":["1337"],"URL":"https:\/\/doi.org\/10.1007\/s10489-018-1337-5","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2018,11,5]]},"assertion":[{"value":"5 November 2018","order":1,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}