{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,27]],"date-time":"2025-06-27T04:14:43Z","timestamp":1750997683730,"version":"3.41.0"},"publisher-location":"Cham","reference-count":66,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319646794"},{"type":"electronic","value":"9783319646800"}],"license":[{"start":{"date-parts":[[2017,1,1]],"date-time":"2017-01-01T00:00:00Z","timestamp":1483228800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017]]},"DOI":"10.1007\/978-3-319-64680-0_6","type":"book-chapter","created":{"date-parts":[[2017,10,31]],"date-time":"2017-10-31T08:37:26Z","timestamp":1509439046000},"page":"135-164","source":"Crossref","is-referenced-by-count":3,"title":["Novel Deep Architectures in Speech Processing"],"prefix":"10.1007","author":[{"given":"John R.","family":"Hershey","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jonathan","family":"Le Roux","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shinji","family":"Watanabe","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Scott","family":"Wisdom","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhuo","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yusuf","family":"Isik","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,7,26]]},"reference":[{"key":"6_CR1","first-page":"297","volume":"5","author":"H. Attias","year":"2003","unstructured":"Attias, H.: New EM algorithms for source separation and deconvolution with a microphone array. In: Proceedings of ICASSP, vol.\u00a05, pp.\u00a0297\u2013300 (2003)","journal-title":"In: Proceedings of ICASSP"},{"key":"6_CR2","unstructured":"Ba, J., Mnih, V., Kavukcuoglu, K.: Multiple object recognition with visual attention (2014). arXiv:1412.7755"},{"key":"6_CR3","unstructured":"Bahdanau, D., Cho, K., Bengio, Y.: Neural machine translation by jointly learning to align and translate (2014). arXiv:1409.0473"},{"issue":"1","key":"6_CR4","doi-asserted-by":"crossref","first-page":"121","DOI":"10.1214\/06-BA104","volume":"1","author":"D.M. Blei","year":"2006","unstructured":"Blei, D.M., Jordan, M.I.: Variational inference for Dirichlet process mixtures. Bayesian Anal. 1(1), 121\u2013144 (2006)","journal-title":"Bayesian Anal."},{"key":"6_CR5","doi-asserted-by":"crossref","DOI":"10.7551\/mitpress\/1486.001.0001","volume-title":"Auditory Scene Analysis: The Perceptual Organization of Sound","author":"A.S. Bregman","year":"1990","unstructured":"Bregman, A.S.: Auditory Scene Analysis: The Perceptual Organization of Sound. MIT Press, Cambridge (1990)"},{"issue":"1","key":"6_CR6","doi-asserted-by":"crossref","first-page":"235","DOI":"10.1007\/s10479-007-0176-2","volume":"153","author":"B. Colson","year":"2007","unstructured":"Colson, B., Marcotte, P., Savard, G.: An overview of bilevel optimization. Ann. Oper. Res. 153(1), 235\u2013256 (2007)","journal-title":"Ann. Oper. Res."},{"key":"6_CR7","doi-asserted-by":"crossref","unstructured":"Domke, J.: Parameter learning with truncated message-passing. In: IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp.\u00a02937\u20132943 (2011)","DOI":"10.1109\/CVPR.2011.5995320"},{"issue":"10","key":"6_CR8","doi-asserted-by":"crossref","first-page":"2454","DOI":"10.1109\/TPAMI.2013.31","volume":"35","author":"J. Domke","year":"2013","unstructured":"Domke, J.: Learning graphical model parameters with approximate marginal inference. IEEE Trans. Pattern Anal. Mach. Intell. 35(10), 2454 (2013)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"7","key":"6_CR9","doi-asserted-by":"crossref","first-page":"1830","DOI":"10.1109\/TASL.2010.2050716","volume":"18","author":"N. Duong","year":"2010","unstructured":"Duong, N., Vincent, E., Gribonval, R.: Under-determined reverberant audio source separation using a full-rank spatial covariance model. IEEE Trans. Audio Speech Lang. Process. 18(7), 1830\u20131840 (2010)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"6_CR10","first-page":"2529","volume":"4","author":"J. Eggert","year":"2004","unstructured":"Eggert, J., K\u00f6rner, E.: Sparse coding and NMF. In: Proceedings of Neural Networks, vol.\u00a04, pp.\u00a02529\u20132533 (2004)","journal-title":"In: Proceedings of Neural Networks"},{"key":"6_CR11","volume-title":"Phase-sensitive and recognition-boosted speech separation using deep recurrent neural networks","author":"H. Erdogan","year":"2015","unstructured":"Erdogan, H., Hershey, J.R., Watanabe, S., Le\u00a0Roux, J.: Phase-sensitive and recognition-boosted speech separation using deep recurrent neural networks. In: Proceedings of ICASSP (2015)"},{"issue":"3","key":"6_CR12","doi-asserted-by":"crossref","first-page":"793","DOI":"10.1162\/neco.2008.04-08-771","volume":"21","author":"C. F\u00e9votte","year":"2009","unstructured":"F\u00e9votte, C., Bertin, N., Durrieu, J.L.: Nonnegative matrix factorization with the Itakura\u2013Saito divergence: with application to music analysis. Neural Comput. 21(3), 793\u2013830 (2009)","journal-title":"Neural Comput."},{"issue":"3","key":"6_CR13","doi-asserted-by":"crossref","first-page":"381","DOI":"10.1109\/34.990138","volume":"24","author":"M.A.T. Figueiredo","year":"2002","unstructured":"Figueiredo, M.A.T., Jain, A.K.: Unsupervised learning of finite mixture models. IEEE Trans. Pattern Anal. Mach. Intell. 24(3), 381\u2013396 (2002)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"6_CR14","unstructured":"Goodfellow, I.J., Mirza, M., Courville, A., Bengio, Y.: Multi-prediction deep Boltzmann machines. In: Advances in Neural Information Processing Systems, pp.\u00a0548\u2013556 (2013)"},{"key":"6_CR15","unstructured":"Goodfellow, I.J., Warde-Farley, D., Mirza, M., Courville, A., Bengio, Y.: Maxout networks (2013). arXiv:1302.4389"},{"key":"6_CR16","unstructured":"Gregor, K., LeCun, Y.: Learning fast approximations of sparse coding. In: ICML, pp.\u00a0399\u2013406 (2010)"},{"issue":"1","key":"6_CR17","doi-asserted-by":"crossref","first-page":"158","DOI":"10.1109\/TASL.2009.2024731","volume":"18","author":"E. Habets","year":"2010","unstructured":"Habets, E., Benesty, J., Cohen, I., Gannot, S., Dmochowski, J.: New insights into the MVDR beamformer in room acoustics. IEEE Trans. Audio Speech Lang. Process. 18(1), 158\u2013170 (2010)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"6_CR18","unstructured":"Hershey, J.R.: Perceptual inference in generative models. Ph.D. thesis, University of California, San Diego (2005)"},{"key":"6_CR19","unstructured":"Hershey, J.R., Le\u00a0Roux, J., Weninger, F.: Deep unfolding: model-based inspiration of novel deep architectures (2014). arXiv:1409.2574"},{"key":"6_CR20","unstructured":"Hershey, J.R., Chen, Z., Le Roux, J., Watanabe, S.: Deep clustering: discriminative embeddings for segmentation and separation (2015). arXiv:1508.04306"},{"key":"6_CR21","volume-title":"Deep clustering: discriminative embeddings for segmentation and separation","author":"J.R. Hershey","year":"2016","unstructured":"Hershey, J.R., Chen, Z., Le Roux, J., Watanabe, S.: Deep clustering: discriminative embeddings for segmentation and separation. In: Proceedings of ICASSP (2016)"},{"key":"6_CR22","volume-title":"Speech acoustic modeling from raw multichannel waveforms","author":"Y. Hoshen","year":"2015","unstructured":"Hoshen, Y., Weiss, R.J., Wilson, K.W.: Speech acoustic modeling from raw multichannel waveforms. In: Proceedings of ICASSP (2015)"},{"key":"6_CR23","doi-asserted-by":"crossref","unstructured":"Huang, P.S., Kim, M., Hasegawa-Johnson, M., Smaragdis, P.: Deep learning for monaural speech separation. In: Proceedings of ICASSP, pp.\u00a01562\u20131566 (2014)","DOI":"10.1109\/ICASSP.2014.6853860"},{"key":"6_CR24","unstructured":"Huang, P.S., Kim, M., Hasegawa-Johnson, M., Smaragdis, P.: Joint optimization of masks and deep recurrent neural networks for monaural source separation (2015). arXiv:1502.04149"},{"key":"6_CR25","volume-title":"Single-channel multi-speaker separation using deep clustering","author":"Y. Isik","year":"2016","unstructured":"Isik, Y., Le Roux, J., Chen, Z., Watanabe, S., Hershey, J.R.: Single-channel multi-speaker separation using deep clustering. In: Proceedings of ISCA Interspeech (2016)"},{"issue":"2","key":"6_CR26","doi-asserted-by":"crossref","first-page":"183","DOI":"10.1023\/A:1007665907178","volume":"37","author":"M.I. Jordan","year":"1999","unstructured":"Jordan, M.I., Ghahramani, Z., Jaakkola, T.S., Saul, L.K.: An introduction to variational methods for graphical models. Mach. Learn. 37(2), 183\u2013233 (1999)","journal-title":"Mach. Learn."},{"key":"6_CR27","unstructured":"Kaiser, L., Sutskever, I.: Neural GPUs learn algorithms (2015). arXiv:1511.08228"},{"key":"6_CR28","volume-title":"The REVERB challenge: a common evaluation framework for dereverberation and recognition of reverberant speech","author":"K. Kinoshita","year":"2013","unstructured":"Kinoshita, K., Delcroix, M., Yoshioka, T., Nakatani, T., Habets, E., Haeb-Umbach, R., Leutnant, V., Sehr, A., Kellermann, W., Maas, R.: The REVERB challenge: a common evaluation framework for dereverberation and recognition of reverberant speech. In: Proceedings of WASPAA (2013)"},{"key":"6_CR29","unstructured":"Kreutz-Delgado, K.: The complex gradient operator and the CR-calculus (2009). arXiv:0906.4835"},{"key":"6_CR30","volume-title":"Deep NMF for speech enhancement","author":"J. Roux Le","year":"2015","unstructured":"Le\u00a0Roux, J., Hershey, J.R., Weninger, F.J.: Deep NMF for speech enhancement. In: Proceedings of ICASSP (2015)"},{"key":"6_CR31","unstructured":"Lee, D.D., Seung, H.S.: Algorithms for non-negative matrix factorization. In: NIPS, pp.\u00a0556\u2013562 (2001)"},{"key":"6_CR32","volume-title":"Mean field networks","author":"Y. Li","year":"2014","unstructured":"Li, Y., Zemel, R.: Mean field networks. In: Learning Tractable Probabilistic Models (2014)"},{"issue":"4","key":"6_CR33","doi-asserted-by":"crossref","first-page":"745","DOI":"10.1109\/TASLP.2014.2304637","volume":"22","author":"J. Li","year":"2014","unstructured":"Li, J., Deng, L., Gong, Y., Haeb-Umbach, R.: An overview of noise-robust automatic speech recognition. IEEE\/ACM Trans. Audio Speech Lang. Process. 22(4), 745\u2013777 (2014)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"4","key":"6_CR34","doi-asserted-by":"crossref","first-page":"791","DOI":"10.1109\/TPAMI.2011.156","volume":"34","author":"J. Mairal","year":"2012","unstructured":"Mairal, J., Bach, F., Ponce, J.: Task-driven dictionary learning. IEEE Trans. Pattern Anal. Mach. Intell. 34(4), 791\u2013804 (2012)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"2","key":"6_CR35","doi-asserted-by":"crossref","first-page":"382","DOI":"10.1109\/TASL.2009.2029711","volume":"18","author":"M.I. Mandel","year":"2010","unstructured":"Mandel, M.I., Weiss, R.J., Ellis, D.P.: Model-based expectation-maximization source separation and localization. IEEE Trans. Audio Speech Lang. Process. 18(2), 382\u2013394 (2010)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"6_CR36","volume-title":"Vision: A Computational Investigation into the Human Representation and Processing of Visual Information","author":"D. Marr","year":"1982","unstructured":"Marr, D.: Vision: A Computational Investigation into the Human Representation and Processing of Visual Information. H. Freeman, San Francisco (1982)"},{"key":"6_CR37","unstructured":"Mnih, V., Heess, N., Graves, A., et\u00a0al.: Recurrent models of visual attention. In: Advances in Neural Information Processing Systems, pp.\u00a02204\u20132212 (2014)"},{"key":"6_CR38","doi-asserted-by":"crossref","unstructured":"Narayanan, A., Wang, D.: Ideal ratio mask estimation using deep neural networks for robust speech recognition. In: Proceedings of ICASSP, pp.\u00a07092\u20137096 (2013)","DOI":"10.1109\/ICASSP.2013.6639038"},{"key":"6_CR39","volume-title":"Probabilistic Reasoning in Intelligent Systems: Networks of Plausible Inference","author":"J. Pearl","year":"1988","unstructured":"Pearl, J.: Probabilistic Reasoning in Intelligent Systems: Networks of Plausible Inference. Morgan Kaufmann, San Francisco (1988)"},{"key":"6_CR40","volume-title":"The Kaldi speech recognition toolkit","author":"D. Povey","year":"2011","unstructured":"Povey, D., Ghoshal, A., Boulianne, G., Burget, L., Glembek, O., Goel, N., Hannemann, M., Motlicek, P., Qian, Y., Schwarz, P., Silovsky, J., Stemmer, G., Vesely, K.: The Kaldi speech recognition toolkit. In: Proceedings of ASRU (2011)"},{"key":"6_CR41","doi-asserted-by":"crossref","unstructured":"Robinson, T., Fransen, J., Pye, D., Foote, J., Renals, S.: WSJCAM0: a British English speech corpus for large vocabulary continuous speech recognition. In: Proceedings of ICASSP, pp.\u00a081\u201384 (1995)","DOI":"10.1109\/ICASSP.1995.479278"},{"key":"6_CR42","unstructured":"Romera-Paredes, B., Torr, P.H.: Recurrent instance segmentation (2015). arXiv:1511.08250"},{"key":"6_CR43","doi-asserted-by":"crossref","unstructured":"Ross, S., Munoz, D., Hebert, M., Bagnell, J.A.: Learning message-passing inference machines for structured prediction. In: IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp.\u00a02737\u20132744 (2011)","DOI":"10.1109\/CVPR.2011.5995724"},{"issue":"5","key":"6_CR44","doi-asserted-by":"crossref","first-page":"489","DOI":"10.1109\/TSA.2004.832988","volume":"12","author":"M.L. Seltzer","year":"2004","unstructured":"Seltzer, M.L., Raj, B., Stern, R.M.: Likelihood-maximizing beamforming for robust hands-free speech recognition. IEEE Trans. Audio Speech Process. 12(5), 489\u2013498 (2004)","journal-title":"IEEE Trans. Audio Speech Process."},{"key":"6_CR45","doi-asserted-by":"crossref","unstructured":"Seltzer, M.L., Yu, D., Wang, Y.: An investigation of deep neural networks for noise robust speech recognition. In: Proceedings of ICASSP, pp.\u00a07398\u20137402 (2013)","DOI":"10.1109\/ICASSP.2013.6639100"},{"key":"6_CR46","unstructured":"Shental, N., Zomet, A., Hertz, T., Weiss, Y.: Pairwise clustering and graphical models. In: Advances in Neural Information Processing Systems, pp.\u00a0185\u2013192 (2004)"},{"key":"6_CR47","doi-asserted-by":"crossref","unstructured":"Smaragdis, P., Raj, B., Shashanka, M.: Supervised and semi-supervised separation of sounds from single-channel mixtures. In: Proceedings of ICA, pp.\u00a0414\u2013421 (2007)","DOI":"10.1007\/978-3-540-74494-8_52"},{"issue":"9","key":"6_CR48","doi-asserted-by":"crossref","first-page":"1913","DOI":"10.1109\/TASL.2013.2263137","volume":"21","author":"M. Souden","year":"2013","unstructured":"Souden, M., Araki, S., Kinoshita, K., Nakatani, T., Sawada, H.: A multichannel MMSE-based framework for speech source separation and noise reduction. IEEE Trans. Audio Speech Lang. Process. 21(9), 1913\u20131928 (2013)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"6_CR49","unstructured":"Sprechmann, P., Litman, R., Yakar, T.B., Bronstein, A.M., Sapiro, G.: Supervised sparse analysis and synthesis operators. In: NIPS, pp.\u00a0908\u2013916 (2013)"},{"key":"6_CR50","volume-title":"Supervised non-Euclidean sparse NMF via bilevel optimization with applications to speech enhancement","author":"P. Sprechmann","year":"2014","unstructured":"Sprechmann, P., Bronstein, A.M., Sapiro, G.: Supervised non-Euclidean sparse NMF via bilevel optimization with applications to speech enhancement. In: Proceedings of HSCMA (2014)"},{"key":"6_CR51","unstructured":"Stoyanov, V., Ropson, A., Eisner, J.: Empirical risk minimization of graphical model parameters given approximate inference, decoding, and model structure. In: International Conference on Artificial Intelligence and Statistics, pp.\u00a0725\u2013733 (2011)"},{"issue":"9","key":"6_CR52","doi-asserted-by":"crossref","first-page":"1120","DOI":"10.1109\/LSP.2014.2325781","volume":"21","author":"P. Swietojanski","year":"2014","unstructured":"Swietojanski, P., Ghoshal, A., Renals, S.: Convolutional neural networks for distant speech recognition. IEEE Signal Process. Lett. 21(9), 1120\u20131124 (2014)","journal-title":"IEEE Signal Process. Lett."},{"issue":"4","key":"6_CR53","doi-asserted-by":"crossref","first-page":"1462","DOI":"10.1109\/TSA.2005.858005","volume":"14","author":"E. Vincent","year":"2006","unstructured":"Vincent, E., Gribonval, R., F\u00e9votte, C.: Performance measurement in blind audio source separation. IEEE Trans. Audio Speech Lang. Process. 14(4), 1462\u20131469 (2006)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"6_CR54","doi-asserted-by":"crossref","unstructured":"Vincent, E., Barker, J., Watanabe, S., Le\u00a0Roux, J., Nesta, F., Matassoni, M.: The second \u2018CHiME\u2019 speech separation and recognition challenge: datasets, tasks and baselines. In: Proceedings of ICASSP, pp.\u00a0126\u2013130 (2013)","DOI":"10.1109\/ICASSP.2013.6637622"},{"issue":"7","key":"6_CR55","doi-asserted-by":"crossref","first-page":"2313","DOI":"10.1109\/TIT.2005.850091","volume":"51","author":"M.J. Wainwright","year":"2005","unstructured":"Wainwright, M.J., Jaakkola, T.S., Willsky, A.S.: A new class of upper bounds on the log partition function. IEEE Trans. Inf. Theory 51(7), 2313\u20132335 (2005)","journal-title":"IEEE Trans. Inf. Theory"},{"issue":"12","key":"6_CR56","doi-asserted-by":"crossref","first-page":"1849","DOI":"10.1109\/TASLP.2014.2352935","volume":"22","author":"Y. Wang","year":"2014","unstructured":"Wang, Y., Narayanan, A., Wang, D.: On training targets for supervised speech separation. IEEE\/ACM IEEE Trans. Audio Speech Lang. Process. 22(12), 1849\u20131858 (2014)","journal-title":"IEEE\/ACM IEEE Trans. Audio Speech Lang. Process."},{"key":"6_CR57","doi-asserted-by":"crossref","unstructured":"Weiss, Y.: Comparing the mean field method and belief propagation for approximate inference in MRFs. In: Advanced Mean Field Methods Theory and Practice, pp.\u00a0229\u2013240 (2001)","DOI":"10.7551\/mitpress\/1100.003.0019"},{"key":"6_CR58","volume-title":"Discriminative NMF and its application to single-channel source separation","author":"F. Weninger","year":"2014","unstructured":"Weninger, F., Le Roux, J., Hershey, J.R., Watanabe, S.: Discriminative NMF and its application to single-channel source separation. In: Proceedings of ISCA Interspeech (2014)"},{"key":"6_CR59","doi-asserted-by":"crossref","unstructured":"Weninger, F., Erdogan, H., Watanabe, S., Vincent, E., Le\u00a0Roux, J., Hershey, J.R., Schuller, B.: Speech enhancement with LSTM recurrent neural networks and its application to noise-robust ASR. In: Latent Variable Analysis and Signal Separation (LVA), pp.\u00a091\u201399 (2015)","DOI":"10.1007\/978-3-319-22482-4_11"},{"key":"6_CR60","doi-asserted-by":"crossref","unstructured":"Wisdom, S., Hershey, J.R., Le Roux, J., Watanabe, S.: Deep unfolding for multichannel source separation: supplementary materials. http:\/\/www.merl.com\/demos\/deep-MCGMM (2015)","DOI":"10.1109\/ICASSP.2016.7471649"},{"key":"6_CR61","doi-asserted-by":"crossref","unstructured":"Wisdom, S., Hershey, J., Le\u00a0Roux, J., Watanabe, S.: Deep unfolding for multichannel source separation. In: Proceedings of ICASSP, pp.\u00a0121\u2013125 (2016)","DOI":"10.1109\/ICASSP.2016.7471649"},{"issue":"1","key":"6_CR62","doi-asserted-by":"crossref","first-page":"65","DOI":"10.1109\/LSP.2013.2291240","volume":"21","author":"Y. Xu","year":"2014","unstructured":"Xu, Y., Du, J., Dai, L.R., Lee, C.H.: An experimental study on speech enhancement based on deep neural networks. IEEE Signal Process. Lett. 21(1), 65\u201368 (2014)","journal-title":"IEEE Signal Process. Lett."},{"key":"6_CR63","volume-title":"Bilevel sparse models for polyphonic music transcription","author":"T.B. Yakar","year":"2013","unstructured":"Yakar, T.B., Litman, R., Sprechmann, P., Bronstein, A., Sapiro, G.: Bilevel sparse models for polyphonic music transcription. In: Proceedings of ISMIR (2013)"},{"issue":"7","key":"6_CR64","doi-asserted-by":"crossref","first-page":"2282","DOI":"10.1109\/TIT.2005.850085","volume":"51","author":"J.S. Yedidia","year":"2005","unstructured":"Yedidia, J.S., Freeman, W.T., Weiss, Y.: Constructing free-energy approximations and generalized belief propagation algorithms. IEEE Trans. Inf. Theory 51(7), 2282\u20132312 (2005)","journal-title":"IEEE Trans. Inf. Theory"},{"key":"6_CR65","unstructured":"Yu, D., Kolb\u00e6k, M., Tan, Z.H., Jensen, J.: Permutation invariant training of deep models for speaker-independent multi-talker speech separation (2016). arXiv:1607.00325"},{"key":"6_CR66","volume-title":"Improving deep neural network acoustic models using generalized maxout networks","author":"X. Zhang","year":"2014","unstructured":"Zhang, X., Trmal, J., Povey, D., Khudanpur, S.: Improving deep neural network acoustic models using generalized maxout networks. In: Proceedings of ICASSP (2014)"}],"container-title":["New Era for Robust Speech Recognition"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-64680-0_6","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,26]],"date-time":"2025-06-26T20:15:59Z","timestamp":1750968959000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-64680-0_6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017]]},"ISBN":["9783319646794","9783319646800"],"references-count":66,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-64680-0_6","relation":{},"subject":[],"published":{"date-parts":[[2017]]}}}