{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T07:27:05Z","timestamp":1740122825355,"version":"3.37.3"},"reference-count":43,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2018,9,5]],"date-time":"2018-09-05T00:00:00Z","timestamp":1536105600000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2018,12]]},"DOI":"10.1007\/s10772-018-9550-5","type":"journal-article","created":{"date-parts":[[2018,9,5]],"date-time":"2018-09-05T04:53:56Z","timestamp":1536123236000},"page":"837-849","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Improvement in monaural speech separation using sparse non-negative tucker decomposition"],"prefix":"10.1007","volume":"21","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9254-8986","authenticated-orcid":false,"given":"Yash Vardhan","family":"Varshney","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Prashant","family":"Upadhyaya","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zia Ahmad","family":"Abbasi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Musiur Raza","family":"Abidi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Omar","family":"Farooq","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2018,9,5]]},"reference":[{"key":"9550_CR1","doi-asserted-by":"crossref","unstructured":"Anastasakos, T., McDonough, J., & Makhoul, J. (1997). Speaker adaptive training: A maximum likelihood approach to speaker normalization. In IEEE international conference on acoustics, speech, and signal processing (pp.\u00a01043\u20131046).","DOI":"10.1109\/ICASSP.1997.596119"},{"key":"9550_CR2","doi-asserted-by":"crossref","unstructured":"Bavkar, S. (2013). PCA based single channel speech enhancement method for highly noisy environment. In Advances in computing, communications and informatics (ICACCI) (pp. 1103\u20131107).","DOI":"10.1109\/ICACCI.2013.6637331"},{"key":"9550_CR3","doi-asserted-by":"crossref","unstructured":"Bertin, N., F\u00e9votte, C., & Badeau, R. (2009). A tempering approach for Itakura-Saito non-negative matrix factorization. With application to music transcription. In Proceedings of ICASSP, IEEE international conference on acoustics, speech and signal processing (pp. 1545\u20131548).","DOI":"10.1109\/ICASSP.2009.4959891"},{"key":"9550_CR4","doi-asserted-by":"publisher","first-page":"1307","DOI":"10.1007\/s13042-017-0645-0","volume":"9","author":"MR Bouguelia","year":"2018","unstructured":"Bouguelia, M. R., Nowaczyk, S., Santosh, K. C., & Verikas, A. (2018). Agreeing to disagree: active learning with noisy labels without crowdsourcing. International Journal of Machine Learning and Cybernetics, 9, 1307\u20131319. https:\/\/doi.org\/10.1007\/s13042-017-0645-0 .","journal-title":"International Journal of Machine Learning and Cybernetics"},{"key":"9550_CR5","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.csl.2009.02.006","volume":"24","author":"M Cooke","year":"2010","unstructured":"Cooke, M., Hershey, J. R., & Rennie, S. J. (2010). Monaural speech separation and recognition challenge. Computer Speech & Language, 24, 1\u201315. https:\/\/doi.org\/10.1016\/j.csl.2009.02.006 .","journal-title":"Computer Speech & Language"},{"key":"9550_CR6","doi-asserted-by":"publisher","first-page":"35","DOI":"10.1007\/978-3-319-73059-2_4","volume-title":"Direction of arrival estimation and localization of multi-speech sources","author":"N Dey","year":"2018","unstructured":"Dey, N., & Ashour, A. S. (2018a). Applied examples and applications of localization and tracking problem of multiple speech sources. In Direction of arrival estimation and localization of multi-speech sources (pp.\u00a035\u201348). Cham: Springer."},{"key":"9550_CR7","doi-asserted-by":"publisher","first-page":"49","DOI":"10.1007\/978-3-319-73059-2_5","volume-title":"Direction of arrival estimation and localization of multi-speech sources","author":"N Dey","year":"2018","unstructured":"Dey, N., & Ashour, A. S. (2018b). Challanges and future perspectives in speech-sources direction of arrival estimation and localization. In Direction of arrival estimation and localization of multi-speech sources (pp.\u00a049\u201352). Cham: Springer."},{"key":"9550_CR8","doi-asserted-by":"publisher","unstructured":"F\u00e9votte, C. (2011). Majorization-minization algorithm for smooth Itakuro-Saito non-negative matrix factorization. Compute 1980\u20131983. https:\/\/doi.org\/10.1109\/ICASSP.2011.5946898 .","DOI":"10.1109\/ICASSP.2011.5946898"},{"key":"9550_CR9","doi-asserted-by":"publisher","first-page":"793","DOI":"10.1162\/neco.2008.04-08-771","volume":"21","author":"C F\u00e9votte","year":"2009","unstructured":"F\u00e9votte, C., Bertin, N., & Durrieu, J.-L. (2009). Nonnegative matrix factorization with the Itakura-Saito divergence: With application to music analysis. Neural Computation, 21, 793\u2013830. https:\/\/doi.org\/10.1162\/neco.2008.04-08-771 .","journal-title":"Neural Computation"},{"key":"9550_CR10","unstructured":"F\u00e9votte, C., Gribonval, R., & Vincent, E. (2005). BSS EVAL Toolbox User Guide. Tech Rep 1706, IRISA."},{"key":"9550_CR11","doi-asserted-by":"publisher","first-page":"75","DOI":"10.1006\/csla.1998.0043","volume":"12","author":"MJF Gales","year":"1998","unstructured":"Gales, M. J. F. (1998). Maximum likelihood linear transformations for HMM-based speech recognition. Computer Speech and Language, 12, 75\u201398. https:\/\/doi.org\/10.1006\/csla.1998.0043 .","journal-title":"Computer Speech and Language"},{"key":"9550_CR12","unstructured":"Garofolo, J., Lamel, L., & Fisher, W., et al. (1988). Getting started with the DARPA TIMIT CD-ROM: An acoustic phonetic continuous speech database. National Institute of Standards and Technology (NIST), Gaithersburg, MD, USA."},{"key":"9550_CR13","doi-asserted-by":"crossref","unstructured":"Guan, N., Lan, L., & Tao, D., et al. (2014). Transductive nonnegative matrix factorization for semi-supervised high-performance speech separation. In Proceedings of ICASSP, IEEE international conference on acoustics, speech and signal processing (pp\u00a02534\u20132538).","DOI":"10.1109\/ICASSP.2014.6854057"},{"key":"9550_CR14","doi-asserted-by":"publisher","first-page":"1457","DOI":"10.1109\/ICMLC.2011.6016966","volume":"5","author":"PO Hoyer","year":"2004","unstructured":"Hoyer, P. O. (2004). Non-negative matrix factorization with sparseness constraints. Journal of Machine Learning Research, 5, 1457\u20131469. https:\/\/doi.org\/10.1109\/ICMLC.2011.6016966 .","journal-title":"Journal of Machine Learning Research"},{"key":"9550_CR15","unstructured":"ITU. (2001). Perceptual evaluation of speech quality (PESQ), an objective method for end-to-end speech quality assessment of narrowband telephone networks and speech codecs. In ITU-T recommendation (pp.\u00a01\u201332)."},{"key":"9550_CR16","unstructured":"Jolliffe, I. T. (2002). Principal component analysis (2nd ed.). Berlin: Springer"},{"key":"9550_CR17","doi-asserted-by":"crossref","unstructured":"Khademian, M., & Mehdi, M. (2016). Monaural multi-talker speech recognition using factorial speech processing models. 1\u201328.","DOI":"10.1016\/j.specom.2018.01.007"},{"key":"9550_CR18","doi-asserted-by":"publisher","unstructured":"Kim, Y.-D. & Choi, S. (2007). Nonnegative tucker decomposition. 1\u20138. https:\/\/doi.org\/10.1109\/CVPR.2007.383405 .","DOI":"10.1109\/CVPR.2007.383405"},{"key":"9550_CR19","doi-asserted-by":"crossref","unstructured":"Kolda, T. G. (2006) Multilinear operators for higher-order decompositions, SANDIA Report SAND2006-2081.","DOI":"10.2172\/923081"},{"key":"9550_CR20","doi-asserted-by":"publisher","first-page":"788","DOI":"10.1038\/44565","volume":"401","author":"DD Lee","year":"1999","unstructured":"Lee, D. D., & Seung, H. S. (1999). Learning the parts of objects by non-negative matrix factorization. Nature, 401, 788\u2013791. https:\/\/doi.org\/10.1038\/44565 .","journal-title":"Nature"},{"key":"9550_CR21","unstructured":"Lef, A., & Bach, F. (2011). Online algorithms for nonnegative matrix factorization with the Itakura-Saito divergence to cite this version: online algorithms for nonnegative matrix factorization with the Itakura-Saito divergence."},{"key":"9550_CR22","doi-asserted-by":"publisher","first-page":"1589","DOI":"10.1109\/TNN.2007.895831","volume":"18","author":"C-J Lin","year":"2007","unstructured":"Lin, C.-J. (2007). On the convergence of multiplicative update for nonnegative matrix factorization. IEEE Transactions on Neural Networks and Learning Systems, 18, 1589\u20131596.","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"9550_CR23","doi-asserted-by":"publisher","first-page":"649","DOI":"10.1016\/j.patcog.2011.05.015","volume":"45","author":"J Liu","year":"2012","unstructured":"Liu, J., Liu, J., Wonka, P., & Ye, J. (2012). Sparse non-negative tensor factorization using columnwise coordinate descent. Pattern Recognition, 45, 649\u2013656.","journal-title":"Pattern Recognition"},{"key":"9550_CR24","doi-asserted-by":"crossref","unstructured":"Mallat, S. (1998) A wavelet tour of signal processing: the sparse way (3rd ed.). Cambridge: Academic Press.","DOI":"10.1016\/B978-012466606-1\/50008-8"},{"key":"9550_CR25","doi-asserted-by":"publisher","first-page":"25","DOI":"10.1007\/s11634-014-0192-4","volume":"11","author":"A Mirzal","year":"2017","unstructured":"Mirzal, A. (2017). NMF versus ICA for blind source separation. Advances in Data Analysis and Classification, 11, 25\u201348. https:\/\/doi.org\/10.1007\/s11634-014-0192-4 .","journal-title":"Advances in Data Analysis and Classification"},{"key":"9550_CR26","unstructured":"M\u00f8rup, M., & Hansen, L. K. (2009) Tuning pruning in sparse non-negative matrix factorization. In European signal processing conference (pp. 1923\u20131927)."},{"key":"9550_CR27","doi-asserted-by":"publisher","DOI":"10.1007\/s10772-018-9525-6","author":"H Mukherjee","year":"2018","unstructured":"Mukherjee, H., Obaidullah, S. M., & Santosh, K. C., et al. (2018). Line spectral frequency-based features and extreme learning machine for voice activity detection from audio signal. International Journal of Speech Technology. https:\/\/doi.org\/10.1007\/s10772-018-9525-6 .","journal-title":"International Journal of Speech Technology"},{"key":"9550_CR28","doi-asserted-by":"publisher","first-page":"982","DOI":"10.1049\/el:19990676","volume":"35","author":"H-M Park","year":"1999","unstructured":"Park, H.-M., Jung, H.-Y., Lee, T.-W., & Lee, S.-Y. (1999). Subband-based blind signal separation for noisy speech recognition. Electronics Letters, 35, 982\u2013984. https:\/\/doi.org\/10.1049\/el:19991358 .","journal-title":"Electronics Letters"},{"key":"9550_CR29","unstructured":"Pl\u00e1tek, O. (2014). Automatic speech recognition using Kaldi. Charles University in Prague."},{"key":"9550_CR30","doi-asserted-by":"publisher","unstructured":"Povey, D., Ghoshal, A., Boulianne, G., et al. (2011). The Kaldi speech recognition toolkit. In IEEE workshop on automatic speech recognition and understanding (pp. 1\u20134). https:\/\/doi.org\/10.1017\/CBO9781107415324.004 .","DOI":"10.1017\/CBO9781107415324.004"},{"key":"9550_CR31","doi-asserted-by":"publisher","first-page":"569","DOI":"10.1109\/18.119724","volume":"38","author":"O Rioul","year":"1992","unstructured":"Rioul, O., & Duhamel, P. (1992). Fast algorithms for discrete and continuous wavelet transforms. IEEE Transactions on Information Theory, 38, 569\u2013586. https:\/\/doi.org\/10.1109\/18.119724 .","journal-title":"IEEE Transactions on Information Theory"},{"key":"9550_CR32","doi-asserted-by":"crossref","unstructured":"Schmidt, M., Winther, O., & Hansen, L. K. (2009). Bayesian non-negative matrix factorization. In Independent component analysis and signal separation (pp. 540\u2013547).","DOI":"10.1007\/978-3-642-00599-2_68"},{"key":"9550_CR33","unstructured":"Stern, R. M. (2003). Signal separation motivated by human auditory perception: Applications to automatic speech recognition. In NSF symposium on speech separation."},{"key":"9550_CR34","doi-asserted-by":"publisher","first-page":"2125","DOI":"10.1109\/TASL.2011.2114881","volume":"19","author":"CH Taal","year":"2011","unstructured":"Taal, C. H., Hendriks, R. C., Heusdens, R., & Jensen, J. (2011). An algorithm for intelligibility prediction of time\u2014Frequency weighted noisy speech. IEEE Transactions on Audio, Speech, and Language Processing, 19, 2125\u20132136.","journal-title":"IEEE Transactions on Audio, Speech, and Language Processing"},{"key":"9550_CR35","doi-asserted-by":"crossref","unstructured":"Upadhyaya, P., Mittal, S. K., Varshney, Y. V., et al. (2017) Speaker adaptive model for hindi speech using Kaldi speech recognition toolkit. In International conference on multimedia, signal processing and communication technologies (IMPACT) (pp.\u00a0222\u2013226).","DOI":"10.1109\/MSPCT.2017.8364009"},{"key":"9550_CR36","doi-asserted-by":"publisher","first-page":"247","DOI":"10.1016\/0167-6393(93)90095-3","volume":"12","author":"A Varga","year":"1993","unstructured":"Varga, A., & Steeneken, H. J. M. (1993). Assessment for automatic speech recognition:{II}. {NOISEX-92}: A database and an experiment to study the effct of additive noise on speech recognition systems. Speech Communication, 12, 247\u2013251.","journal-title":"Speech Communication"},{"key":"9550_CR37","doi-asserted-by":"crossref","unstructured":"Varshney, Y. V., Abbasi, Z. A., Abidi, M. R., & Farooq, O. (2017a). Variable sparsity regularization factor based SNMF for monaural speech separation. In 2017 40th international conference on telecommunications and signal processing, TSP 2017.","DOI":"10.1109\/TSP.2017.8076001"},{"key":"9550_CR38","doi-asserted-by":"publisher","first-page":"287","DOI":"10.1515\/aoa-2017-0031","volume":"42","author":"YV Varshney","year":"2017","unstructured":"Varshney, Y. V., Abbasi, Z. A., Abidi, M. R., & Farooq, O. (2017b). Frequency selection based separation of speech signals with reduced computational time using sparse NMF. Archives of Acoustics, 42, 287\u2013295. https:\/\/doi.org\/10.1515\/aoa-2017-0031 .","journal-title":"Archives of Acoustics"},{"key":"9550_CR39","first-page":"1462","volume":"14","author":"E Vincent","year":"2006","unstructured":"Vincent, E., Gribonval, R., & F\u00b4evotte, C. (2006). Performance measurement in blind audio source separation. IEEE Transactions on Audio, Speech, and Language Processing Institute of Electrical and Electronics Engineers, 14, 1462\u20131469.","journal-title":"IEEE Transactions on Audio, Speech, and Language Processing Institute of Electrical and Electronics Engineers"},{"key":"9550_CR40","doi-asserted-by":"publisher","unstructured":"Virtanen, T., Cemgil, A. T., & Godsill, S. (2008). Bayesian extensions to non-negative matrix factorisation for audio signal modelling. In Proceedings of ICASSP, IEEE international conference on acoustics, speech, and signal processing (pp. 1825\u20131828). https:\/\/doi.org\/10.1109\/ICASSP.2008.4517987 .","DOI":"10.1109\/ICASSP.2008.4517987"},{"key":"9550_CR41","unstructured":"Young, S., Hain, T., & Woodland, P., et al. (2002). The HTK book (for version 3.2.1). Cambridge: Cambridge University Engineering Department."},{"key":"9550_CR42","unstructured":"Yuan, Z., Yang, Z., & Oja, E. (2007) Projective nonnegative matrix factorization: Sparseness, orthogonality, and clustering. Helsinki University of Technology 1\u201314."},{"key":"9550_CR43","doi-asserted-by":"publisher","first-page":"4990","DOI":"10.1109\/TIP.2015.2478396","volume":"24","author":"G Zhou","year":"2015","unstructured":"Zhou, G., Cichocki, A., Zhao, Q., & Xie, S. (2015). Efficient nonnegative tucker decompositions: Algorithms and uniqueness. IEEE Transactions on Image Processing, 24, 4990\u20135003. https:\/\/doi.org\/10.1109\/TIP.2015.2478396 .","journal-title":"IEEE Transactions on Image Processing"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10772-018-9550-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-018-9550-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-018-9550-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,10,23]],"date-time":"2019-10-23T13:23:36Z","timestamp":1571837016000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10772-018-9550-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,9,5]]},"references-count":43,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2018,12]]}},"alternative-id":["9550"],"URL":"https:\/\/doi.org\/10.1007\/s10772-018-9550-5","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"type":"print","value":"1381-2416"},{"type":"electronic","value":"1572-8110"}],"subject":[],"published":{"date-parts":[[2018,9,5]]},"assertion":[{"value":"1 February 2018","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 August 2018","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 September 2018","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}