{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T21:48:28Z","timestamp":1751406508965,"version":"3.37.3"},"reference-count":30,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2016,7,25]],"date-time":"2016-07-25T00:00:00Z","timestamp":1469404800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100003399","name":"Science and Technology Commission of Shanghai Municipality","doi-asserted-by":"publisher","award":["14YF1409300"],"award-info":[{"award-number":["14YF1409300"]}],"id":[{"id":"10.13039\/501100003399","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003395","name":"Shanghai Municipal Education Commission","doi-asserted-by":"publisher","award":["ZZshsf14026"],"award-info":[{"award-number":["ZZshsf14026"]}],"id":[{"id":"10.13039\/501100003395","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2016,9]]},"DOI":"10.1007\/s10772-016-9355-3","type":"journal-article","created":{"date-parts":[[2016,7,25]],"date-time":"2016-07-25T13:25:09Z","timestamp":1469453109000},"page":"623-630","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":8,"title":["Improvements on self-adaptive voice activity detector for telephone data"],"prefix":"10.1007","volume":"19","author":[{"given":"Haoran","family":"Wei","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanhua","family":"Long","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hongwei","family":"Mao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2016,7,25]]},"reference":[{"key":"9355_CR1","doi-asserted-by":"crossref","first-page":"64","DOI":"10.1109\/35.620527","volume":"35","author":"A Benyassine","year":"1997","unstructured":"Benyassine, A., Schlomot, E., & Su, H. Y. (1997). ITU-T recommendation g729 annex b: A silence compression scheme for use with g729 optimized for v. 70 digital simultaneous voice and data applications. IEEE Communications Magazine, 35, 64\u201373.","journal-title":"IEEE Communications Magazine"},{"key":"9355_CR2","doi-asserted-by":"crossref","first-page":"430","DOI":"10.1155\/S1110865704310024","volume":"2004","author":"F Bimbot","year":"2004","unstructured":"Bimbot, F., et al. (2004). A tutorial on text-independent speaker verification. EURASIP Journal on Applied Signal Processing, 2004, 430\u2013451.","journal-title":"EURASIP Journal on Applied Signal Processing"},{"key":"9355_CR3","unstructured":"Brummer, N., et al. (2010). ABC system description for NIST SRE 2010. NIST 2010 Speaker Recognition Evaluation (pp. 1\u201320)."},{"key":"9355_CR4","doi-asserted-by":"crossref","first-page":"1979","DOI":"10.1109\/TASL.2007.902499","volume":"15","author":"L Burget","year":"2007","unstructured":"Burget, L., et al. (2007). Analysis of feature extraction and channel compensation in a GMM speaker recognition system. IEEE Transactions on Audio, Speech and Language Processing, 15, 1979\u20131986.","journal-title":"IEEE Transactions on Audio, Speech and Language Processing"},{"key":"9355_CR5","doi-asserted-by":"crossref","unstructured":"Burlick, M., et al. (2013). On the improvement of multimodal voice activity detection. In Interspeech (pp. 685\u2013689).","DOI":"10.21437\/Interspeech.2013-194"},{"key":"9355_CR6","unstructured":"ETSI. (1999). Detector V A. for adaptive multi-rate (AMR) speech traffic channels."},{"key":"9355_CR7","unstructured":"ETSI. (2002). Speech processing, transmission and quality aspects (STQ); distributed speech recognition; advanced front-end feature extraction algorithm; compression algorithms. ETSI ES 201-108 Recommendation."},{"key":"9355_CR8","doi-asserted-by":"crossref","first-page":"369","DOI":"10.1109\/ICASSP.1989.266442","volume":"1","author":"DK Freeman","year":"1989","unstructured":"Freeman, D. K., Cosier, G., & Southcott, C. B. (1989). The voice activity detector for the Pan-European digital cellular mobile telephone service. International Conference on. IEEE Acoustics, Speech, and Signal Processing, 1, 369\u2013372.","journal-title":"International Conference on. IEEE Acoustics, Speech, and Signal Processing"},{"key":"9355_CR9","doi-asserted-by":"crossref","unstructured":"Ganapathy, S., et al. (2011). Multi-layer perceptron based speech activity detection for speaker verification. In IEEE Workshop on Applications of Signal Processing to Audio and Acoustics (WASPAA 2011) (pp. 321\u2013324).","DOI":"10.1109\/ASPAA.2011.6082323"},{"key":"9355_CR10","doi-asserted-by":"crossref","first-page":"600","DOI":"10.1109\/TASL.2010.2052803","volume":"19","author":"PK Ghosh","year":"2011","unstructured":"Ghosh, P. K., et al. (2011). Robust voice activity detection using long-term signal variability. IEEE Transactions on Audio, Speech and Language Processing, 19, 600\u2013613.","journal-title":"IEEE Transactions on Audio, Speech and Language Processing"},{"key":"9355_CR11","unstructured":"Hautamaki, V., et al. (2007). Improving speaker verification by periodicity based voice activity detection. In Proceeding of the 12th International Conference on Speech and Computer (SPECOM 2007) (pp. 645\u2013650)."},{"key":"9355_CR12","doi-asserted-by":"crossref","unstructured":"Heese, F., et al. (2015). Speech-codebook based soft voice activity detection. In ICASSP (pp. 4335\u20134339).","DOI":"10.1109\/ICASSP.2015.7178789"},{"key":"9355_CR13","doi-asserted-by":"crossref","unstructured":"Huijbregts, M., Wooters, C., & Ordelman. R. (2007). Filtering the unknown: Speech activity detection in heterogeneous video collections. In Interspeech (pp. 2925\u20132928).","DOI":"10.21437\/Interspeech.2007-729"},{"key":"9355_CR14","unstructured":"Kenny, P., Ouellet, P., & Senoussaoui, M. (2010). The CRIM system for the 2010 nist speaker recognition evaluation. In Proceeding of the NIST 2010 Speaker Recognition Evaluation."},{"key":"9355_CR15","doi-asserted-by":"crossref","first-page":"12","DOI":"10.1016\/j.specom.2009.08.009","volume":"52","author":"T Kinnunen","year":"2010","unstructured":"Kinnunen, T., & Li, H. (2010). An overview of text-independent speaker recognition: From features to supervectors. Speech Communication, 52, 12\u201340.","journal-title":"Speech Communication"},{"key":"9355_CR16","doi-asserted-by":"crossref","unstructured":"Kinnunen, T., & Rajan, P. (2013). A practical, self-adaptive voice activity detector for speaker verification with noisy telephone and microphone data. In ICASSP (pp. 7229\u20137233).","DOI":"10.1109\/ICASSP.2013.6639066"},{"issue":"7","key":"9355_CR17","doi-asserted-by":"crossref","first-page":"2026","DOI":"10.1109\/TASL.2011.2109379","volume":"19","author":"I McCowan","year":"2011","unstructured":"McCowan, I., et al. (2011). The delta-phase spectrum with application to voice activity detection and speaker recognition. Audio, Speech, and Language Processing, IEEE Transactions on, 19(7), 2026\u20132038.","journal-title":"Audio, Speech, and Language Processing, IEEE Transactions on"},{"key":"9355_CR18","unstructured":"NIST Multimodal Information Group (2006). NIST speaker recognition evaluation training set LDC2011S09. Web download. Philadelphia: Linguistic Data Consortium, 2011. https:\/\/catalog.ldc.upenn.edu\/LDC2011S09 . Accessed 16 Nov 2011."},{"issue":"2","key":"9355_CR19","doi-asserted-by":"crossref","first-page":"297","DOI":"10.1002\/j.1538-7305.1975.tb02840.x","volume":"54","author":"LR Rabiner","year":"1975","unstructured":"Rabiner, L. R., & Sambur, M. R. (1975). An algorithm for determining the endpoints of isolated utterances. Bell System Technical Journal, 54(2), 297\u2013315.","journal-title":"Bell System Technical Journal"},{"issue":"2","key":"9355_CR20","doi-asserted-by":"crossref","first-page":"266","DOI":"10.1109\/LSP.2003.821762","volume":"11","author":"J Ram\u00edrez","year":"2004","unstructured":"Ram\u00edrez, J., Segura, J. C., & Ben\u00edtez, C. (2004). A new Kullback-Leibler VAD for speech recognition in noise. IEEE Signal Processing Letters, 11(2), 266\u2013269.","journal-title":"IEEE Signal Processing Letters"},{"issue":"6","key":"9355_CR21","doi-asserted-by":"crossref","first-page":"1119","DOI":"10.1109\/TSA.2005.853212","volume":"13","author":"J Ram\u00edrez","year":"2005","unstructured":"Ram\u00edrez, J., Segura, J. C., & Ben\u00edtez, C. (2005). An effective subband OSF-based VAD with noise reduction for robust speech recognition. IEEE Transactions on Speech and Audio Processing, 13(6), 1119\u20131129.","journal-title":"IEEE Transactions on Speech and Audio Processing"},{"key":"9355_CR22","unstructured":"Sahidullah, M., & Saha, G. (2012). Comparison of speech activity detection techniques for speaker recognition. arXiv preprint arXiv:1210.0297 ."},{"key":"9355_CR23","doi-asserted-by":"crossref","unstructured":"Sangwan, A., Chiranth, M. C., & Jamadagni, H. S. (2002). VAD techniques for real-time speech transmission on the internet. High Speed Networks and Multimedia Communications 5th IEEE International Conference on IEEE (pp. 46\u201350).","DOI":"10.1109\/HSNMC.2002.1032545"},{"key":"9355_CR24","doi-asserted-by":"crossref","unstructured":"Saon, G., Thomas, S., & Soltau, H. (2013). The IBM speech activity detection system for the DARPA RATS program. In Interspeech (pp. 3497\u20133501).","DOI":"10.21437\/Interspeech.2013-264"},{"key":"9355_CR25","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1109\/97.736233","volume":"6","author":"J Sohn","year":"1999","unstructured":"Sohn, J., Kim, N. S., & Sung, W. (1999). A statistical model-based voice activity detection. IEEE Signal Processing Letters, 6, 1\u20133.","journal-title":"IEEE Signal Processing Letters"},{"key":"9355_CR26","doi-asserted-by":"crossref","unstructured":"Sun, H., Nwe, T. L., Ma, B., & Li, H. (2009). Speaker diarization for meeting room audio. In Proceeding of the Interspeech (pp. 900\u2013903).","DOI":"10.21437\/Interspeech.2009-271"},{"key":"9355_CR27","doi-asserted-by":"crossref","unstructured":"Thomas, S., Saon, G., & Segbroeck, M. V. (2015). Improvements to the IBM speech activity detection system for the DARPA RATS program. International Conference on Acoustics, Speech, and Signal Processing IEEE (pp. 4500\u20134504).","DOI":"10.1109\/ICASSP.2015.7178822"},{"key":"9355_CR28","doi-asserted-by":"crossref","unstructured":"Ye, J., et al. (2013). Incremental acoustic subspace learning for voice activity detection using harmonicity-based features. In Interspeech (pp. 695\u2013699).","DOI":"10.21437\/Interspeech.2013-196"},{"key":"9355_CR29","doi-asserted-by":"crossref","unstructured":"Yu, H. B., & Mak, M. W. (2011). Comparison of voice activity detectors for interview speech in NIST speaker recognition evaluation. In Interspeech (pp. 2353\u20132356).","DOI":"10.21437\/Interspeech.2011-61"},{"key":"9355_CR30","doi-asserted-by":"crossref","unstructured":"Zhang, X. L., & Wu, J. (2013). Denoising deep neural networks based voice activity detection. Acoustics, Speech and Signal Processing (ICASSP) (pp. 853\u2013857).","DOI":"10.1109\/ICASSP.2013.6637769"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-016-9355-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10772-016-9355-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-016-9355-3","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,3]],"date-time":"2022-07-03T19:13:12Z","timestamp":1656875592000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10772-016-9355-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016,7,25]]},"references-count":30,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2016,9]]}},"alternative-id":["9355"],"URL":"https:\/\/doi.org\/10.1007\/s10772-016-9355-3","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"type":"print","value":"1381-2416"},{"type":"electronic","value":"1572-8110"}],"subject":[],"published":{"date-parts":[[2016,7,25]]}}}