{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,15]],"date-time":"2026-04-15T13:20:28Z","timestamp":1776259228621,"version":"3.50.1"},"publisher-location":"Cham","reference-count":22,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783319569031","type":"print"},{"value":"9783319569048","type":"electronic"}],"license":[{"start":{"date-parts":[[2017,8,30]],"date-time":"2017-08-30T00:00:00Z","timestamp":1504051200000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2018]]},"DOI":"10.1007\/978-3-319-56904-8_16","type":"book-chapter","created":{"date-parts":[[2017,8,29]],"date-time":"2017-08-29T08:22:49Z","timestamp":1503994969000},"page":"161-170","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["Convolutional Neural Networks with 3-D Kernels for Voice Activity Detection in a Multiroom Environment"],"prefix":"10.1007","author":[{"given":"Paolo","family":"Vecchiotti","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fabio","family":"Vesperini","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Emanuele","family":"Principi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Stefano","family":"Squartini","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Francesco","family":"Piazza","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,8,30]]},"reference":[{"key":"16_CR1","unstructured":"Abad, A., Matos, M., Meinedo, H., Astudillo, R.F., Trancoso, I.: The L2F system for the EVALITA-2014 speech activity detection challenge in domestic environments. In: Proceedings of EVALITA, pp. 147\u2013152 (2014)"},{"key":"16_CR2","unstructured":"Cristoforetti, L., Ravanelli, M., Omologo, M., Sosi, A., Abad, A., Hagm\u00fcller, M., Maragos, P.: The DIRHA simulated corpus. In: Proceedings of LREC, vol.\u00a05. Reykjavik, Iceland (2014)"},{"key":"16_CR3","doi-asserted-by":"crossref","unstructured":"Ferroni, G., Bonfigli, R., Principi, E., Squartini, S., Piazza, F.: A deep neural network approach for voice activity detection in multi-room domestic scenarios. In: Proceedings of IJCNN, pp. 1\u20138. Killarney, Ireland (2015)","DOI":"10.1109\/IJCNN.2015.7280510"},{"key":"16_CR4","doi-asserted-by":"crossref","unstructured":"Gemmeke, J.F., Ons, B., Tessema, N., Van Hamme, H., van\u00a0de Loo, J., De Pauw, G., Daelemans, W., Huyghe, J., Derboven, J., Vuegen, L., Van Den Broeck, B., Karsmakers, P., Vanrumste, B.: Self-taught assistive vocal interfaces: an overview of the ALADIN project. In: Proceedings of Interspeech, pp. 2039\u20132043. Lyon, France (2013)","DOI":"10.21437\/Interspeech.2013-483"},{"key":"16_CR5","unstructured":"Giannoulis, P., Tsiami, A., Rodomagoulakis, I., Katsamanis, A., Potamianos, G., Maragos, P.: The Athena-RC system for speech activity detection and speaker localization in the DIRHA smart home. In: Proceedings of HSCMA, 2014, pp. 167\u2013171. Florence, Italy (2014)"},{"key":"16_CR6","doi-asserted-by":"crossref","unstructured":"Hussain, A., Chetouani, M., Squartini, S., Bastari, A., Piazza, F.: Nonlinear Speech Enhancement: An Overview, pp. 217\u2013248. Springer, Berlin (2007)","DOI":"10.1007\/978-3-540-71505-4_12"},{"key":"16_CR7","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. In: Advances in Neural Information Processing Systems, pp. 1097\u20131105 (2012)"},{"issue":"11","key":"16_CR8","doi-asserted-by":"publisher","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y LeCun","year":"1998","unstructured":"LeCun, Y., Bottou, L., Bengio, Y., Haffner, P.: Gradient-based learning applied to document recognition. Proc. IEEE 86(11), 2278\u20132324 (1998)","journal-title":"Proc. IEEE"},{"key":"16_CR9","doi-asserted-by":"crossref","DOI":"10.1201\/b14529","volume-title":"Speech Enhancement: Theory and Practice","author":"PC Loizou","year":"2013","unstructured":"Loizou, P.C.: Speech Enhancement: Theory and Practice. CRC Press, Boca Raton, FL (2013)"},{"issue":"1","key":"16_CR10","doi-asserted-by":"publisher","first-page":"114","DOI":"10.1016\/j.patcog.2013.06.029","volume":"47","author":"N Lopes","year":"2014","unstructured":"Lopes, N., Ribeiro, B.: Towards adaptive learning with improved convergence of deep belief networks on graphics processing units. Pattern Recogn. 47(1), 114\u2013127 (2014)","journal-title":"Pattern Recogn."},{"key":"16_CR11","doi-asserted-by":"crossref","unstructured":"McLoughlin, I., Song, Y.: Low frequency ultrasonic voice activity detection using convolutional neural networks. In: Proceedings of Interspeech. Dresden, Germany (2015)","DOI":"10.21437\/Interspeech.2015-519"},{"key":"16_CR12","doi-asserted-by":"crossref","unstructured":"Mohamed, A., Hinton, G., Penn, G.: Understanding how deep belief networks perform acoustic modelling. In: Proceedings of ICASSP, pp. 4273\u20134276. Kyoto, Japan (2012)","DOI":"10.1109\/ICASSP.2012.6288863"},{"key":"16_CR13","unstructured":"Morales-Cordovilla, J.A., Hagmuller, M., Pessentheiner, H., Kubin, G.: Distant speech recognition in reverberant noisy conditions employing a microphone array. In: Proceedings of EUSIPCO, pp. 2380\u20132384. Lisbona, Portugal (2014)"},{"key":"16_CR14","doi-asserted-by":"crossref","unstructured":"Price, R., Iso, K.I., Shinoda, K.: Wise teachers train better DNN acoustic models. Eurasip J. Audio Speech Music Process 2016(1) (2016)","DOI":"10.1186\/s13636-016-0088-7"},{"issue":"13","key":"16_CR15","doi-asserted-by":"publisher","first-page":"5668","DOI":"10.1016\/j.eswa.2015.02.036","volume":"42","author":"E Principi","year":"2015","unstructured":"Principi, E., Squartini, S., Bonfigli, R., Ferroni, G., Piazza, F.: An integrated system for voice command recognition and emergency detection based on audio signals. Expert Syst. Appl. 42(13), 5668\u20135683 (2015)","journal-title":"Expert Syst. Appl."},{"issue":"4","key":"16_CR16","doi-asserted-by":"publisher","first-page":"504","DOI":"10.1007\/s12559-012-9176-x","volume":"5","author":"R Rotili","year":"2013","unstructured":"Rotili, R., Principi, E., Squartini, S., Schuller, B.: A real-time speech enhancement framework in noisy and reverberated acoustic scenarios. Cogn. Comput. 5(4), 504\u2013516 (2013)","journal-title":"Cogn. Comput."},{"key":"16_CR17","doi-asserted-by":"publisher","first-page":"533","DOI":"10.1038\/323533a0","volume":"323","author":"DE Rumelhart","year":"1986","unstructured":"Rumelhart, D.E., Hinton, G.E., Williams, R.J.: Learning representations by back-propagating errors. Nature 323, 533\u2013536 (1986)","journal-title":"Nature"},{"key":"16_CR18","doi-asserted-by":"crossref","unstructured":"Thomas, S., Ganapathy, S., Saon, G., Soltau, H.: Analyzing convolutional neural networks for speech activity detection in mismatched acoustic conditions. In: Proceedings of ICASSP, pp. 2519\u20132523. Florence, Italy (2014)","DOI":"10.1109\/ICASSP.2014.6854054"},{"key":"16_CR19","unstructured":"Ullrich, K., Schl\u00fcter, J., Grill, T.: Boundary detection in music structure analysis using convolutional neural networks. In: Proceedings of ISMIR, pp. 417\u2013422. Taipei, Taiwan (2014)"},{"key":"16_CR20","doi-asserted-by":"crossref","unstructured":"Vacher, M., Caffiau, S., Portet, F., Meillon, B., Roux, C., Elias, E., Lecouteux, B., Chahuara, P.: Evaluation of a context-aware voice interface for ambient assisted living: qualitative user study vs. quantitative system evaluation. ACM Trans. Access. Comput. 7(2), 5:1\u20135:36 (2015)","DOI":"10.1145\/2738047"},{"key":"16_CR21","doi-asserted-by":"crossref","unstructured":"Vesperini, F., Vecchiotti, P., Principi, E., Squartini, S., Piazza, F.: Deep neural networks for multi-room voice activity detection: advancements and comparative evaluation. In: Proceedings of IJCNN, pp. 3391\u20133398. Vancouver, Canada (2016)","DOI":"10.1109\/IJCNN.2016.7727633"},{"key":"16_CR22","unstructured":"Zhang, X.L., Wang, D.: Boosting contextual information for deep neural network based voice activity detection. IEEE\/ACM Trans. Audio Speech Lang. Process. 24(2), 252\u2013264 (2016)"}],"container-title":["Smart Innovation, Systems and Technologies","Multidisciplinary Approaches to Neural Computing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-56904-8_16","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,8,1]],"date-time":"2022-08-01T21:48:50Z","timestamp":1659390530000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-56904-8_16"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017,8,30]]},"ISBN":["9783319569031","9783319569048"],"references-count":22,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-56904-8_16","relation":{},"ISSN":["2190-3018","2190-3026"],"issn-type":[{"value":"2190-3018","type":"print"},{"value":"2190-3026","type":"electronic"}],"subject":[],"published":{"date-parts":[[2017,8,30]]},"assertion":[{"value":"30 August 2017","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}