{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,29]],"date-time":"2025-12-29T18:56:12Z","timestamp":1767034572995,"version":"3.37.3"},"reference-count":50,"publisher":"Springer Science and Business Media LLC","issue":"11","license":[{"start":{"date-parts":[[2018,7,19]],"date-time":"2018-07-19T00:00:00Z","timestamp":1531958400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Comput &amp; Applic"],"published-print":{"date-parts":[[2019,11]]},"DOI":"10.1007\/s00521-018-3623-x","type":"journal-article","created":{"date-parts":[[2018,7,19]],"date-time":"2018-07-19T14:48:18Z","timestamp":1532011698000},"page":"7989-8002","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":12,"title":["Robust feature extraction and uncertainty estimation based on attractor dynamics in cyclic deep denoising autoencoders"],"prefix":"10.1007","volume":"31","author":[{"given":"Amir Hossein","family":"Hadjahmadi","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2379-4427","authenticated-orcid":false,"given":"Mohammad Mehdi","family":"Homayounpour","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2018,7,19]]},"reference":[{"key":"3623_CR1","doi-asserted-by":"crossref","unstructured":"Graves A, Fern\u00e1ndez S, Gomez F (2006) Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks. In: Proceedings of the international conference on machine learning, ICML 2006, pp 369\u2013376","DOI":"10.1145\/1143844.1143891"},{"key":"3623_CR2","doi-asserted-by":"crossref","unstructured":"Bahdanau D, Chorowski J, Serdyuk D et al (2016) End-to-end attention-based large vocabulary speech recognition. In: IEEE international conference on acoustics, speech, and signal processing, pp 4945\u20134949","DOI":"10.1109\/ICASSP.2016.7472618"},{"key":"3623_CR3","doi-asserted-by":"publisher","first-page":"161","DOI":"10.1109\/89.260359","volume":"2","author":"S Renals","year":"1994","unstructured":"Renals S, Morgan N, Bourlard H et al (1994) Connectionist probability estimators in HMM speech recognition. IEEE Trans Speech Audio Process 2:161\u2013174. https:\/\/doi.org\/10.1109\/89.260359","journal-title":"IEEE Trans Speech Audio Process"},{"key":"3623_CR4","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","volume":"29","author":"G Hinton","year":"2012","unstructured":"Hinton G, Deng L, Yu D et al (2012) Deep neural networks for acoustic modeling in speech recognition: the shared views of four research groups. IEEE Signal Process Mag 29:82\u201397. https:\/\/doi.org\/10.1109\/MSP.2012.2205597","journal-title":"IEEE Signal Process Mag"},{"key":"3623_CR5","doi-asserted-by":"crossref","unstructured":"Hermansky H, Ellis DP, Sharma S (2000) Tandem connectionist feature extraction for conventional HMM systems. In: IEEE international conference on acoustics, speech, and signal processing, pp 1635\u20131638","DOI":"10.1109\/ICASSP.2000.862024"},{"key":"3623_CR6","doi-asserted-by":"crossref","unstructured":"Grezl F, Karafiat M, Kontar S, Cernocky J (2007) Probabilistic and Bottle-Neck features for LVCSR of meetings. In: 2007 IEEE international conference on acoustics, speech and signal processing, pp 757\u2013760","DOI":"10.1109\/ICASSP.2007.367023"},{"key":"3623_CR7","first-page":"3371","volume":"11","author":"P Vincent","year":"2010","unstructured":"Vincent P, Larochelle H, Lajoie I et al (2010) Stacked denoising autoencoders: learning useful representations in a deep network with a local denoising criterion. J Mach Learn Res 11:3371\u20133408","journal-title":"J Mach Learn Res"},{"key":"3623_CR8","doi-asserted-by":"publisher","first-page":"7","DOI":"10.1109\/TASLP.2014.2364452","volume":"23","author":"Y Xu","year":"2015","unstructured":"Xu Y, Du J, Dai LR, Lee CH (2015) A regression approach to speech enhancement based on deep neural networks. IEEE ACM Trans Audio Speech Lang Process 23:7\u201319. https:\/\/doi.org\/10.1109\/TASLP.2014.2364452","journal-title":"IEEE ACM Trans Audio Speech Lang Process"},{"key":"3623_CR9","doi-asserted-by":"crossref","unstructured":"Feng X, Zhang Y, Glass J (2014) Speech feature denoising and dereverberation via deep autoencoders for noisy reverberant speech recognition. In: 2014 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 1759\u20131763","DOI":"10.1109\/ICASSP.2014.6853900"},{"key":"3623_CR10","doi-asserted-by":"publisher","first-page":"12","DOI":"10.1186\/s13636-015-0056-7","volume":"2015","author":"Z Zhang","year":"2015","unstructured":"Zhang Z, Wang L, Kai A et al (2015) Deep neural network-based bottleneck feature and denoising autoencoder-based dereverberation for distant-talking speaker identification. EURASIP J Audio Speech Music Process 2015:12. https:\/\/doi.org\/10.1186\/s13636-015-0056-7","journal-title":"EURASIP J Audio Speech Music Process"},{"key":"3623_CR11","doi-asserted-by":"publisher","first-page":"1527","DOI":"10.1162\/neco.2006.18.7.1527","volume":"18","author":"GE Hinton","year":"2006","unstructured":"Hinton GE, Osindero S, Teh Y-W (2006) A fast learning algorithm for deep belief nets. Neural Comput 18:1527\u20131554. https:\/\/doi.org\/10.1162\/neco.2006.18.7.1527","journal-title":"Neural Comput"},{"key":"3623_CR12","doi-asserted-by":"publisher","first-page":"504","DOI":"10.1126\/science.1127647","volume":"313","author":"GE Hinton","year":"2006","unstructured":"Hinton GE, Salakhutdinov RR (2006) Reducing the dimensionality of data with neural networks. Science 313:504\u2013507. https:\/\/doi.org\/10.1126\/science.1127647","journal-title":"Science"},{"key":"3623_CR13","doi-asserted-by":"crossref","unstructured":"Thomas S, Seltzer ML, Church K, Hermansky H (2013) Deep neural network features and semi-supervised training for low resource speech recognition. In: 2013 IEEE international conference on acoustics, speech and signal processing, pp 6704\u20136708","DOI":"10.1109\/ICASSP.2013.6638959"},{"key":"3623_CR14","doi-asserted-by":"crossref","unstructured":"Ninomiya H, Kitaoka N, Tamura S et al (2015) Integration of deep bottleneck features for audio-visual speech recognition. In: Sixteenth annual conference of the international speech communication association, pp 563\u2013567","DOI":"10.21437\/Interspeech.2015-204"},{"key":"3623_CR15","doi-asserted-by":"crossref","unstructured":"Wu Z, Valentini-Botinhao C, Watts O, King S (2015) Deep neural networks employing multi-task learning and stacked bottleneck features for speech synthesis. In: 2015 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 4460\u20134464","DOI":"10.1109\/ICASSP.2015.7178814"},{"key":"3623_CR16","doi-asserted-by":"crossref","unstructured":"Yu C, Ogawa A, Delcroix M et al (2015) Robust i-vector extraction for neural network adaptation in noisy environment. In: Sixteenth annual conference of the international speech communication association, pp 2854\u20132857","DOI":"10.21437\/Interspeech.2015-600"},{"key":"3623_CR17","doi-asserted-by":"crossref","unstructured":"Vinyals O, Ravuri SV (2011) Comparing multilayer perceptron to Deep Belief Network Tandem features for robust ASR. In: 2011 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 4596\u20134599","DOI":"10.1109\/ICASSP.2011.5947378"},{"key":"3623_CR18","doi-asserted-by":"crossref","unstructured":"Yu D, Seltzer ML (2011) Improved bottleneck features using pretrained deep neural networks. In: Twelfth annual conference of the international speech communication association, pp 237\u2013240","DOI":"10.21437\/Interspeech.2011-91"},{"key":"3623_CR19","doi-asserted-by":"publisher","first-page":"3881","DOI":"10.4249\/scholarpedia.3881","volume":"3","author":"Y Bengio","year":"2008","unstructured":"Bengio Y (2008) Neural net language models. Scholarpedia 3:3881. https:\/\/doi.org\/10.4249\/scholarpedia.3881","journal-title":"Scholarpedia"},{"key":"3623_CR20","doi-asserted-by":"publisher","first-page":"459","DOI":"10.1007\/s00521-009-0328-1","volume":"19","author":"B Makki","year":"2010","unstructured":"Makki B, Hosseini MN, Seyyedsalehi SA (2010) An evolving neural network to perform dynamic principal component analysis. Neural Comput Appl 19:459\u2013463. https:\/\/doi.org\/10.1007\/s00521-009-0328-1","journal-title":"Neural Comput Appl"},{"key":"3623_CR21","doi-asserted-by":"crossref","unstructured":"Sainath TN, Kingsbury B, Ramabhadran B (2012) Auto-encoder bottleneck features using deep belief networks. In: 2012 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 4153\u20134156","DOI":"10.1109\/ICASSP.2012.6288833"},{"key":"3623_CR22","doi-asserted-by":"crossref","unstructured":"Gehring J, Miao Y, Metze F, Waibel A (2013) Extracting deep bottleneck features using stacked auto-encoders. In: 2013 IEEE international conference on acoustics, speech and signal processing, pp 3377\u20133381","DOI":"10.1109\/ICASSP.2013.6638284"},{"key":"3623_CR23","doi-asserted-by":"crossref","unstructured":"Zhang Y, Chuangsuwanich E, Glass J (2014) Extracting deep neural network bottleneck features using low-rank matrix factorization. In: 2014 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 185\u2013189","DOI":"10.1109\/ICASSP.2014.6853583"},{"key":"3623_CR24","doi-asserted-by":"crossref","unstructured":"Seltzer ML, Yu D, Wang Y (2013) An investigation of deep neural networks for noise robust speech recognition. In: 2013 IEEE international conference on acoustics, speech and signal processing, pp 7398\u20137402","DOI":"10.1109\/ICASSP.2013.6639100"},{"key":"3623_CR25","doi-asserted-by":"crossref","first-page":"436","DOI":"10.21437\/Interspeech.2013-130","volume":"2013","author":"X Lu","year":"2013","unstructured":"Lu X, Tsao Y, Matsuda S, Hori C (2013) Speech enhancement based on deep denoising autoencoder. Proc Interspeech 2013:436\u2013440","journal-title":"Proc Interspeech"},{"key":"3623_CR26","doi-asserted-by":"publisher","first-page":"7","DOI":"10.1109\/TASLP.2014.2364452","volume":"23","author":"Y Xu","year":"2015","unstructured":"Xu Y, Du J, Dai L-R, Lee C-H (2015) A regression approach to speech enhancement based on deep neural networks. IEEE ACM Trans Audio Speech Lang Process TASLP 23:7\u201319. https:\/\/doi.org\/10.1109\/TASLP.2014.2364452","journal-title":"IEEE ACM Trans Audio Speech Lang Process TASLP"},{"key":"3623_CR27","doi-asserted-by":"crossref","unstructured":"Feng X, Zhang Y, Glass J (2014) Speech feature denoising and dereverberation via deep autoencoders for noisy reverberant speech recognition. In: 2014 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 1759\u20131763","DOI":"10.1109\/ICASSP.2014.6853900"},{"key":"3623_CR28","doi-asserted-by":"crossref","unstructured":"Heymann J, Haeb-Umbach R, Golik P, Schl\u00fcter R (2015) Unsupervised adaptation of a denoising autoencoder by bayesian feature enhancement for reverberant asr under mismatch conditions. In: IEEE international conference on acoustics, speech, and signal processing, pp 5053\u20135057","DOI":"10.1109\/ICASSP.2015.7178933"},{"key":"3623_CR29","doi-asserted-by":"publisher","first-page":"15","DOI":"10.1016\/j.engappai.2016.12.012","volume":"59","author":"\u0110T Grozdi\u0107","year":"2017","unstructured":"Grozdi\u0107 \u0110T, Jovi\u010di\u0107 ST, Suboti\u0107 M (2017) Whispered speech recognition using deep denoising autoencoder. Eng Appl Artif Intell 59:15\u201322. https:\/\/doi.org\/10.1016\/j.engappai.2016.12.012","journal-title":"Eng Appl Artif Intell"},{"key":"3623_CR30","doi-asserted-by":"crossref","unstructured":"Vincent P, Larochelle H, Bengio Y, Manzagol P-A (2008) Extracting and Composing Robust Features with Denoising Autoencoders. In: Proceedings of the 25th international conference on machine learning. ACM, New York, NY, USA, pp 1096\u20131103","DOI":"10.1145\/1390156.1390294"},{"key":"3623_CR31","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1561\/2200000006","volume":"2","author":"Y Bengio","year":"2009","unstructured":"Bengio Y (2009) Learning deep architectures for AI. Found Trends\u00ae. Mach Learn 2:1\u2013127. https:\/\/doi.org\/10.1561\/2200000006","journal-title":"Mach Learn"},{"key":"3623_CR32","doi-asserted-by":"crossref","unstructured":"Li J, Deng L, Haeb-Umbach R, Gong Y (2015) Robust automatic speech recognition: a bridge to practical applications, 1st edn. Elsevier","DOI":"10.1016\/B978-0-12-802398-3.00001-5"},{"key":"3623_CR33","doi-asserted-by":"publisher","first-page":"175","DOI":"10.1007\/BF00365229","volume":"26","author":"S-I Amari","year":"1977","unstructured":"Amari S-I (1977) Neural theory of association and concept-formation. Biol Cybern 26:175\u2013185. https:\/\/doi.org\/10.1007\/BF00365229","journal-title":"Biol Cybern"},{"key":"3623_CR34","doi-asserted-by":"publisher","first-page":"2554","DOI":"10.1073\/pnas.79.8.2554","volume":"79","author":"JJ Hopfield","year":"1982","unstructured":"Hopfield JJ (1982) Neural networks and physical systems with emergent collective computational abilities. Proc Natl Acad Sci USA 79:2554\u20132558","journal-title":"Proc Natl Acad Sci USA"},{"key":"3623_CR35","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511623257","volume-title":"Modeling brain function\u2014the world of attractor neural networks","author":"DJ Amit","year":"1989","unstructured":"Amit DJ (1989) Modeling brain function\u2014the world of attractor neural networks. Cambridge University Press, New York"},{"key":"3623_CR36","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"238","DOI":"10.1007\/3-540-63508-4_128","volume-title":"Image analysis and processing. ICIAP 1997","author":"DO Gorodnichy","year":"1997","unstructured":"Gorodnichy DO, Reznik AM (1997) Static and dynamic attractors of autoassociative neural networks. In: Del Bimbo A (ed) Image analysis and processing. ICIAP 1997. Lecture Notes in Computer Science, vol. 1311. Springer, Berlin, Heidelberg, pp 238\u2013245. https:\/\/doi.org\/10.1007\/3-540-63508-4_128"},{"key":"3623_CR37","unstructured":"Bengio Y, Yao L, Alain G, Vincent P (2013) Generalized denoising auto-encoders as generative models. In: Proceedings of the 26th international conference on neural information processing systems. Curran Associates Inc., USA, pp 899\u2013907"},{"key":"3623_CR38","doi-asserted-by":"publisher","first-page":"2716","DOI":"10.1016\/j.neucom.2010.12.044","volume":"74","author":"L Dehyadegary","year":"2011","unstructured":"Dehyadegary L, Ali Seyyedsalehi S, Nejadgholi I (2011) Nonlinear enhancement of noisy speech, using continuous attractor dynamics formed in recurrent neural networks. Neurocomputing 74:2716\u20132724. https:\/\/doi.org\/10.1016\/j.neucom.2010.12.044","journal-title":"Neurocomputing"},{"key":"3623_CR39","unstructured":"Droppo J, Acero A, Deng L (2002) Uncertainty decoding with SPLICE for noise robust speech recognition. In: 2002 IEEE international conference on acoustics, speech, and signal processing, pp I-57\u2013I-60"},{"key":"3623_CR40","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-21317-5","volume-title":"Robust speech recognition of uncertain or missing data: theory and applications","author":"D Kolossa","year":"2011","unstructured":"Kolossa D, Haeb-Umbach R (2011) Robust speech recognition of uncertain or missing data: theory and applications. Springer-Verlag, Berlin"},{"key":"3623_CR41","doi-asserted-by":"publisher","first-page":"1023","DOI":"10.1109\/TASL.2013.2244085","volume":"21","author":"RF Astudillo","year":"2013","unstructured":"Astudillo RF, Orglmeister R (2013) Computing MMSE estimates and residual uncertainty directly in the feature domain of ASR using STFT domain speech distortion models. IEEE Trans Audio Speech Lang Process 21:1023\u20131034. https:\/\/doi.org\/10.1109\/TASL.2013.2244085","journal-title":"IEEE Trans Audio Speech Lang Process"},{"key":"3623_CR42","doi-asserted-by":"crossref","unstructured":"Astudillo RF, Abad A, Trancoso I (2014) Accounting for the residual uncertainty of multi-layer perceptron based features. In: 2014 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 6859\u20136863","DOI":"10.1109\/ICASSP.2014.6854929"},{"key":"3623_CR43","first-page":"3561","volume":"2015","author":"AH Abdelaziz","year":"2015","unstructured":"Abdelaziz AH, Watanabe S, Hershey JR et al (2015) Uncertainty propagation through deep neural networks. Proc Interspeech 2015:3561\u20133565","journal-title":"Proc Interspeech"},{"key":"3623_CR44","doi-asserted-by":"publisher","first-page":"30","DOI":"10.1016\/j.csl.2017.06.005","volume":"47","author":"J Novoa","year":"2018","unstructured":"Novoa J, Fredes J, Poblete V, Yoma NB (2018) Uncertainty weighting and propagation in DNN\u2013HMM-based speech recognition. Comput Speech Lang 47:30\u201346. https:\/\/doi.org\/10.1016\/j.csl.2017.06.005","journal-title":"Comput Speech Lang"},{"key":"3623_CR45","unstructured":"Goodfellow I, Yoshua Bengio, Aaron Courville (2016) Deep learning. In: MIT Press. https:\/\/mitpress.mit.edu\/books\/deep-learning . Accessed 23 Oct 2017"},{"key":"3623_CR46","doi-asserted-by":"crossref","first-page":"3110","DOI":"10.21437\/Interspeech.2010-774","volume":"2010","author":"DB Dean","year":"2010","unstructured":"Dean DB, Sridharan S, Vogt RJ, Mason MW (2010) The QUT-NOISE-TIMIT corpus for the evaluation of voice activity detection algorithms. Proc Interspeech 2010:3110\u20133113","journal-title":"Proc Interspeech"},{"key":"3623_CR47","doi-asserted-by":"publisher","first-page":"412","DOI":"10.1109\/TSA.2005.845814","volume":"13","author":"L Deng","year":"2005","unstructured":"Deng L, Droppo J, Acero A (2005) Dynamic compensation of HMM variances using the feature enhancement uncertainty computed from a parametric model of speech distortion. IEEE Trans Speech Audio Process 13:412\u2013421. https:\/\/doi.org\/10.1109\/TSA.2005.845814","journal-title":"IEEE Trans Speech Audio Process"},{"key":"3623_CR48","doi-asserted-by":"publisher","first-page":"2130","DOI":"10.1109\/TASL.2007.901836","volume":"15","author":"S Srinivasan","year":"2007","unstructured":"Srinivasan S, Wang D (2007) Transforming binary uncertainties for robust speech recognition. IEEE Trans Audio Speech Lang Process 15:2130\u20132140. https:\/\/doi.org\/10.1109\/TASL.2007.901836","journal-title":"IEEE Trans Audio Speech Lang Process"},{"key":"3623_CR49","unstructured":"Srinivasan S, Wang D (2006) A supervised learning approach to uncertainty decoding for robust speech recognition. In: 2006 IEEE international conference on acoustics speech and signal processing proceedings, pp I\u2013I"},{"key":"3623_CR50","doi-asserted-by":"crossref","unstructured":"Vinyals O, Ravuri SV, Povey D (2012) Revisiting recurrent neural networks for robust ASR. In: 2012 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 4085\u20134088","DOI":"10.1109\/ICASSP.2012.6288816"}],"container-title":["Neural Computing and Applications"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-018-3623-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s00521-018-3623-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-018-3623-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,8]],"date-time":"2024-07-08T11:44:49Z","timestamp":1720439089000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s00521-018-3623-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,7,19]]},"references-count":50,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2019,11]]}},"alternative-id":["3623"],"URL":"https:\/\/doi.org\/10.1007\/s00521-018-3623-x","relation":{},"ISSN":["0941-0643","1433-3058"],"issn-type":[{"type":"print","value":"0941-0643"},{"type":"electronic","value":"1433-3058"}],"subject":[],"published":{"date-parts":[[2018,7,19]]},"assertion":[{"value":"23 October 2017","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 July 2018","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 July 2018","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Compliance with ethical standards"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}