{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,11]],"date-time":"2026-02-11T18:30:57Z","timestamp":1770834657783,"version":"3.50.1"},"reference-count":46,"publisher":"Springer Science and Business Media LLC","issue":"43-44","license":[{"start":{"date-parts":[[2020,8,26]],"date-time":"2020-08-26T00:00:00Z","timestamp":1598400000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,8,26]],"date-time":"2020-08-26T00:00:00Z","timestamp":1598400000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"name":"National Key Research and Development Program of China","award":["No. 2017YFB1002803"],"award-info":[{"award-number":["No. 2017YFB1002803"]}]},{"DOI":"10.13039\/501100001809","name":"National Nature Science Foundation of China","doi-asserted-by":"crossref","award":["No. U1736206"],"award-info":[{"award-number":["No. U1736206"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"National Nature Science Foundation of China","doi-asserted-by":"crossref","award":["No. 61761044"],"award-info":[{"award-number":["No. 61761044"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100012239","name":"Hubei Province Technological Innovation Major Project","doi-asserted-by":"crossref","award":["No. 2017AAA123"],"award-info":[{"award-number":["No. 2017AAA123"]}],"id":[{"id":"10.13039\/501100012239","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2020,11]]},"DOI":"10.1007\/s11042-020-09419-y","type":"journal-article","created":{"date-parts":[[2020,8,26]],"date-time":"2020-08-26T21:03:42Z","timestamp":1598475822000},"page":"32225-32241","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Single Channel multi-speaker speech Separation based on quantized ratio mask and residual network"],"prefix":"10.1007","volume":"79","author":[{"given":"Shanfa","family":"Ke","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ruimin","family":"Hu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaochen","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tingzhao","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gang","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhongyuan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,8,26]]},"reference":[{"key":"9419_CR1","doi-asserted-by":"crossref","unstructured":"Aihara R, Hanazawa T, Okato Y, et al. (2019). Teacher-student deep clustering for low- delay Single Channel speech Separation[C]\/\/ICASSP 2019-2019 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, 690\u2013694.","DOI":"10.1109\/ICASSP.2019.8682695"},{"issue":"4","key":"9419_CR2","doi-asserted-by":"publisher","first-page":"831","DOI":"10.1109\/TASLP.2017.2789320","volume":"26","author":"Y Bando","year":"2018","unstructured":"Bando Y, Nakamura E, Itakura K, Kawahara T (2018) Bayesian multichannel audio source separation based on integrated source and spatial models. IEEE\/ACM Transac- tions on Audio, Speech, and Language Processing 26(4):831\u2013C846","journal-title":"IEEE\/ACM Transac- tions on Audio, Speech, and Language Processing"},{"key":"9419_CR3","doi-asserted-by":"crossref","unstructured":"Bregman, AS (1990). Auditory scene analysis (The MIT Press, Cambridge, MA), Chap. 1.","DOI":"10.7551\/mitpress\/1486.001.0001"},{"issue":"2","key":"9419_CR4","doi-asserted-by":"publisher","first-page":"287","DOI":"10.1109\/LSP.2016.2514845","volume":"23","author":"TST Chan","year":"2016","unstructured":"Chan TST, Yang YH (2016) Complex and quaternionic principal component pursuit and its application to audio separation[J]. IEEE Signal Processing Letters 23(2):287\u2013291","journal-title":"IEEE Signal Processing Letters"},{"key":"9419_CR5","doi-asserted-by":"crossref","unstructured":"Cherry EC (1953) Some experiments on the recognition of speech,with one and with two ears. The Journal of the acoustical society of America 25(5):975C\u20139979C","DOI":"10.1121\/1.1907229"},{"key":"9419_CR6","doi-asserted-by":"crossref","unstructured":"Dai L, Du J, Tu Y, Lee C (2016) A regression approach to single-channel speech separation via high-resolution deep neural networks, IEEE trans. Audio, speech. Language Process(TASLP) 24(8):1424C\u201311437C","DOI":"10.1109\/TASLP.2016.2558822"},{"key":"9419_CR7","doi-asserted-by":"crossref","unstructured":"Ephrat A, Mosseri I, Lang O, et al. (2018). Looking to listen at the cocktail party: a speaker- independent audio-visual model for speech separation[C]. international conference on computer graphics and interactive techniques, 37(4).","DOI":"10.1145\/3197517.3201357"},{"key":"9419_CR8","doi-asserted-by":"crossref","unstructured":"John S Garofolo, Lori F Lamel, William M Fisher, Jonathan G Fiscus, and David S Pallett (1993). Darpa timit acoustic-phonetic continous speech corpus cd-rom. nist speech disc 1\u20131.1, NASA STI\/Recon technical report n, vol. 93","DOI":"10.6028\/NIST.IR.4930"},{"key":"9419_CR9","doi-asserted-by":"crossref","unstructured":"Gemmeke JF, Virtanen T, Raj B (2013) Active-set newton algorithm for overcomplete non-negative representations of audio. IEEE Trans Audio, Speech, Language Process (TASLP) 21(11):2277C\u201322289C","DOI":"10.1109\/TASL.2013.2263144"},{"key":"9419_CR10","doi-asserted-by":"crossref","unstructured":"Gong Y, Li J, Deng L, Haeb-Umbach R (2014) An overview of noise-robust automatic speech recognition. IEEE TransAudio, Speech, Language Process (TASLP) 22(4):745C\u20137777C","DOI":"10.1109\/TASLP.2014.2304637"},{"key":"9419_CR11","doi-asserted-by":"crossref","unstructured":"Han K, Wang Y, Wang D (2013) Exploring monaural features for classification-based speech segregation. IEEE Trans Audio, Speech, Language Process (TASLP) 21(2):270C\u20132279C","DOI":"10.1109\/TASL.2012.2221459"},{"key":"9419_CR12","unstructured":"M Hasegawa-Johnson P Huang, M Kim and P Smaragdis (2014). Deep learning for monaural speech separation, in acoustics, speech and signal processing (ICASSP), 2014 IEEE international conference on. 2014, pp. 1562C-1566, IEEE"},{"issue":"1","key":"9419_CR13","doi-asserted-by":"publisher","first-page":"122","DOI":"10.1109\/TASL.2012.2215591","volume":"21","author":"K Hu","year":"2013","unstructured":"Hu K, Wang D (2013) An unsupervised approach to cochan- nel speech separation. IEEE Trans Audio, Speech, Language Process (TASLP) 21(1):122\u2013C131","journal-title":"IEEE Trans Audio, Speech, Language Process (TASLP)"},{"issue":"1","key":"9419_CR14","first-page":"122C131","volume":"21","author":"K Hu","year":"2013","unstructured":"Hu K, Wang D (2013) An unsupervised approach to cochannel speech separation. IEEE Trans Audio, Speech, Language Process (TASLP) 21(1):122C131","journal-title":"IEEE Trans Audio, Speech, Language Process (TASLP)"},{"key":"9419_CR15","doi-asserted-by":"crossref","unstructured":"Hummersone, Christopher, Toby Stokes, and Tim Brookes (2014). \u201cOn the ideal ratio mask as the goal of computational auditory scene analysis.\u201d Blind source separation. Springer, Berlin, Heidelberg, 349\u2013368","DOI":"10.1007\/978-3-642-55016-4_12"},{"key":"9419_CR16","doi-asserted-by":"crossref","unstructured":"Hyv\u00e4rinen A, Oja E (2000) Independent component analysis: algorithms and applications. Neural Netw 13(4\u20135):411C\u20134430C","DOI":"10.1016\/S0893-6080(00)00026-5"},{"key":"9419_CR17","doi-asserted-by":"crossref","unstructured":"Isik Y, Roux JL, Chen Z, et al. (2016). Single-channel multi-speaker separation using deep clustering[C]. conference of the international speech communication association: 545\u2013549","DOI":"10.21437\/Interspeech.2016-1176"},{"key":"9419_CR18","doi-asserted-by":"crossref","unstructured":"Ke S, Hu R, Li G, et al. (2019). Multi-speakers speech Separation based on modified attractor points estimation and GMM clustering[C]\/\/2019 IEEE international conference on multimedia and expo (ICME). IEEE, 1414\u20131419.","DOI":"10.1109\/ICME.2019.00245"},{"key":"9419_CR19","doi-asserted-by":"crossref","unstructured":"J Le Roux JR Hershey, Z Chen and S Watanabe (2016). Deep clustering: discriminative embeddings for segmentation and separation, in acoustics, speech and signal processing (ICASSP), 2016 IEEE international conference on. 2016, pp.31-C35, IEEE.","DOI":"10.1109\/ICASSP.2016.7471631"},{"issue":"2","key":"9419_CR20","doi-asserted-by":"publisher","first-page":"370","DOI":"10.1109\/JSTSP.2019.2904183","volume":"13","author":"J Le Roux","year":"2019","unstructured":"Le Roux J, Wichern G, Watanabe S et al (2019) Phasebook and friends: leveraging discrete representations for source separation[J]. IEEE Journal of Selected Topics in Signal Pro- cessing 13(2):370\u2013382","journal-title":"IEEE Journal of Selected Topics in Signal Pro- cessing"},{"issue":"3","key":"9419_CR21","first-page":"645","volume":"27","author":"X Li","year":"2019","unstructured":"Li X, Girin L, Gannot S et al (2019) Multichannel Speech Separation and Enhancement Using the Convolutive Transfer Function[J]. IEEE\/ACM transactions on audio. Speech and Language Processing (TASLP) 27(3):645\u2013659","journal-title":"Speech and Language Processing (TASLP)"},{"issue":"9","key":"9419_CR22","doi-asserted-by":"publisher","first-page":"1315","DOI":"10.1109\/LSP.2018.2853566","volume":"25","author":"R Lu","year":"2018","unstructured":"Lu R, Duan Z, Zhang C (2018) Listen and look: AudioCVisual matching assisted speech source Separation[J]. IEEE Signal Processing Letters 25(9):1315\u20131319","journal-title":"IEEE Signal Processing Letters"},{"key":"9419_CR23","doi-asserted-by":"crossref","unstructured":"Y Luo Z Chen and N Mesgarani (2017). Deep attractor network for single-microphone s- peaker separation, in acoustics, speech and signal processing (ICASSP), 2017 IEEE international conference on. pp. 246C-250, IEEE.","DOI":"10.1109\/ICASSP.2017.7952155"},{"issue":"8","key":"9419_CR24","doi-asserted-by":"publisher","first-page":"1256","DOI":"10.1109\/TASLP.2019.2915167","volume":"27","author":"Y Luo","year":"2019","unstructured":"Luo Y, Mesgarani N (2019) Conv-tasnet: surpassing ideal timeCfrequency magnitude mask- ing for speech separation[J]. IEEE\/ACM transactions on audio, speech, and language processing 27(8):1256\u20131266","journal-title":"IEEE\/ACM transactions on audio, speech, and language processing"},{"issue":"2","key":"9419_CR25","doi-asserted-by":"publisher","first-page":"382","DOI":"10.1109\/TASL.2009.2029711","volume":"18","author":"MI Mandel","year":"2010","unstructured":"Mandel MI, Weiss R, Ellis DP et al (2010) Model-based expectation-maximization source Separation and localization[J]. IEEE Trans Audio Speech Lang Process 18(2):382\u2013394","journal-title":"IEEE Trans Audio Speech Lang Process"},{"key":"9419_CR26","doi-asserted-by":"crossref","unstructured":"Narayanan, Arun, and DeLiang Wang (2013). \u201cIdeal ratio mask estimation using deep neu- ral networks for robust speech recognition.\u201d 2013 IEEE International Conference on Acoustics, Speech and Signal Processing. IEEE","DOI":"10.1109\/ICASSP.2013.6639038"},{"issue":"12","key":"9419_CR27","doi-asserted-by":"publisher","first-page":"1849","DOI":"10.1109\/TASLP.2014.2352935","volume":"22","author":"A Narayanan","year":"2014","unstructured":"Narayanan A, Wang Y, Wang D (2014) On training targets for supervised speech sepa- ration. IEEE\/ACM Transactions on Audio, Speech and Language Processing (TASLP) 22(12):1849\u2013C1858","journal-title":"IEEE\/ACM Transactions on Audio, Speech and Language Processing (TASLP)"},{"key":"9419_CR28","unstructured":"MS.Pedersen (2006). Source separation for hearingaid applications, IMM, Informatik og Matematisk Modelling, DTU"},{"key":"9419_CR29","unstructured":"Colin Raffel, Brian McFee, Eric J. Humphrey, Justin Salamon, Oriol Nieto, Dawen Liang, and Daniel P. W. Ellis (2014). Mir evalA Transparent Implementation of Common MIR Metrics, Proceedings of the 15th International Conference on Music Information Re- trieval, 2014"},{"key":"9419_CR30","doi-asserted-by":"publisher","first-page":"925","DOI":"10.1007\/s11042-013-1398-8","volume":"72.1","author":"FJ Rodrguez-Serrano","year":"2014","unstructured":"Rodrguez-Serrano FJ et al (2014) Monophonic constrained non-negative sparse coding using instrument models for audio separation and transcription of monophonic source-based polyphonic mixtures. Multimed Tools Appl 72.1:925\u2013949","journal-title":"Multimed Tools Appl"},{"key":"9419_CR31","doi-asserted-by":"crossref","unstructured":"A Senior TN Sainath, O Vinyals and H Sak (2015). Convolutional, long short-term mem- ory, fully connected deep neural networks, in acoustics, speech and signal processing (ICASSP),2015 IEEE international conference on. pp. 4580-C4584,IEEE","DOI":"10.1109\/ICASSP.2015.7178838"},{"key":"9419_CR32","doi-asserted-by":"crossref","unstructured":"Smaragdis P, Mohammadiha N, Leijon A (2013) Supervised and unsupervised speech enhancement using nonnegative matrix factorization. IEEE Trans Audio, Speech, Lan- guage Process (TASLP) 21(10):2140C\u201322151C","DOI":"10.1109\/TASL.2013.2270369"},{"key":"9419_CR33","doi-asserted-by":"crossref","unstructured":"Tan Z, Kolb\u00e6k M, Yu D, Jensen J (2017) Multitalker speech separation with utterance-level permutation invariant training of deep recurrent neural networks. IEEE Trans Audio, Speech, Language Process (TASLP) 25(10):1901C\u201311913C","DOI":"10.1109\/TASLP.2017.2726762"},{"key":"9419_CR34","unstructured":"Z Tan D Yu, M Kolb\u00e6k and J Jensen (2017). Permutation invariant training of deep models for speaker-independent multi-talker speech separation, in acoustics, speech and signal processing (ICASSP), 2017 IEEE international conference on. 2017, pp. 241C-245, IEEE"},{"key":"9419_CR35","doi-asserted-by":"crossref","unstructured":"Vasko JL, Carter BL, Healy EW, Delfarah M, Wang D (2017) An algorithm to increase intelligibility for hearing-impaired listeners in the presence of a competing talker. The Journal of the Acoustical Society of America 141(6):4230C\u201344239C","DOI":"10.1121\/1.4984271"},{"issue":"15","key":"9419_CR36","doi-asserted-by":"publisher","first-page":"20129","DOI":"10.1007\/s11042-017-5458-3","volume":"77","author":"R Venkatesan","year":"2018","unstructured":"Venkatesan R, Balaji Ganesh A (2018) Deep recurrent neural networks based binaural speech segregation for the selection of closest target of interest. Multimed Tools Appl 77(15):20129\u201320156","journal-title":"Multimed Tools Appl"},{"key":"9419_CR37","doi-asserted-by":"crossref","unstructured":"Vincent E, Gribonval \u0154, Fevotte \u0106 (2006) Performance measurement in blind audio source separation. IEEE transactions on audio, speech, and language processing 14(4):1462C\u201311469C","DOI":"10.1109\/TSA.2005.858005"},{"issue":"3","key":"9419_CR38","doi-asserted-by":"publisher","first-page":"1066","DOI":"10.1109\/TASL.2006.885253","volume":"15","author":"T Virtanen","year":"2007","unstructured":"Virtanen T (2007) Monaural sound source separation by nonnegative matrix factorization with temporal continuity and sparseness criteria. IEEE Trans Audio, Speech, Language Process(TASLP) 15(3):1066\u2013C1074","journal-title":"IEEE Trans Audio, Speech, Language Process(TASLP)"},{"key":"9419_CR39","doi-asserted-by":"crossref","unstructured":"Wang, DL (2005). On ideal binary mask as the computational goal of auditory scene analysis, in Speech Separation by Humans and Machines, edited by P. Divenyi (Kluwer Academic, Dordrecht), pp. 181C197","DOI":"10.1007\/0-387-22794-6_12"},{"key":"9419_CR40","unstructured":"Zhong-Qiu Wang, Jonathan Le Roux, and John R. Hershey (2018). Alternative objective func- tions for deep clustering, in 2018 IEEE international conference on acoustics, speech and signal processing (ICASSP). pp. 686C-690, IEEE."},{"key":"9419_CR41","unstructured":"Zhong-Qiu Wang, Jonathan Le Roux, DeLiang Wang, and John R. Hershey (2018). End-to- end speech separation with unfolded iterative phase reconstruction, arXiv preprint arX- iv:1804.10204"},{"key":"9419_CR42","doi-asserted-by":"crossref","unstructured":"Wang Z Q, Tan K, Wang D L (2019). Deep learning based phase reconstruction for speak- er separation: a trigonometric perspective[C]\/\/ICASSP 2019-2019 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, 71\u201375.","DOI":"10.1109\/ICASSP.2019.8683231"},{"key":"9419_CR43","doi-asserted-by":"publisher","first-page":"2718","DOI":"10.21437\/Interspeech.2018-1940","volume":"2018","author":"Z Wang","year":"2018","unstructured":"Wang Z, Wang D (2018) Integrating spectral and spatial features for multi-channel s-peaker separation. Proc Interspeech 2018:2718\u2013C2722","journal-title":"Proc Interspeech"},{"key":"9419_CR44","first-page":"483","volume":"24.3","author":"DS Williamson","year":"2015","unstructured":"Williamson DS, Wang Y, Wang DL (2015) Complex ratio masking for monaural speech separation. IEEE\/ACM transactions on audio, speech, and language processing 24.3:483\u2013492","journal-title":"IEEE\/ACM transactions on audio, speech, and language processing"},{"key":"9419_CR45","unstructured":"H Zen K Simonyan O Vinyals A Graves N Kalchbrenner AW Senior A Van Den Oord, S Dieleman and K Kavukcuoglu (2016). Wavenet: A generative model for raw audio., in SSW, p. 125."},{"key":"9419_CR46","doi-asserted-by":"crossref","unstructured":"Zhang X, Wang D (2016) A deep ensemble learning method for monaural speech separation. IEEE Trans Audio, Speech,Language Process (TASLP) 24(5):967C\u20139977C","DOI":"10.1109\/TASLP.2016.2536478"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-020-09419-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-020-09419-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-020-09419-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,8,25]],"date-time":"2021-08-25T23:31:41Z","timestamp":1629934301000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-020-09419-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,8,26]]},"references-count":46,"journal-issue":{"issue":"43-44","published-print":{"date-parts":[[2020,11]]}},"alternative-id":["9419"],"URL":"https:\/\/doi.org\/10.1007\/s11042-020-09419-y","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"value":"1380-7501","type":"print"},{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020,8,26]]},"assertion":[{"value":"15 September 2019","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 June 2020","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 July 2020","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 August 2020","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}