{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T15:11:21Z","timestamp":1784301081796,"version":"3.55.0"},"reference-count":29,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017,3]]},"DOI":"10.1109\/icassp.2017.7952154","type":"proceedings-article","created":{"date-parts":[[2017,6,20]],"date-time":"2017-06-20T21:35:36Z","timestamp":1497994536000},"page":"241-245","source":"Crossref","is-referenced-by-count":561,"title":["Permutation invariant training of deep models for speaker-independent multi-talker speech separation"],"prefix":"10.1109","author":[{"given":"Dong","family":"Yu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Morten","family":"Kolbaek","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zheng-Hua","family":"Tan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jesper","family":"Jensen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref10","author":"ellis","year":"1996","journal-title":"Prediction-driven computational auditory scene analysis"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1037\/11496-005"},{"key":"ref12","article-title":"Single-channel speech separation using sparse non-negative matrix factorization","author":"schmidt","year":"2006","journal-title":"InterSpeech"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2006.876726"},{"key":"ref14","article-title":"Sparse nmf&#x2014;half-baked or well done","author":"le roux","year":"2015","journal-title":"Mitsubishi Electric Research Labs (MERL) Cambridge MA USA Tech Rep no TR2015-023"},{"key":"ref15","first-page":"155","article-title":"Super-human multi-talker speech recognition: the ibm 2006 speech separation challenge system","volume":"12","author":"kristjansson","year":"2006","journal-title":"InterSpeech"},{"key":"ref16","article-title":"Speech recognition using factorial hidden markov models for separation in the feature space","author":"virtanen","year":"2006","journal-title":"InterSpeech"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ASPAA.2007.4393039"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1023\/A:1007425814087"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2014.2352935"},{"key":"ref28","article-title":"An introduction to computational networks and the computational network toolkit","year":"2014","journal-title":"Tech Rep Microsoft Technical Report MSR-TR-2014-112"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2012.2205597"},{"key":"ref27","year":"0","journal-title":"Akustiske databaser for dansk"},{"key":"ref3","doi-asserted-by":"crossref","first-page":"437","DOI":"10.21437\/Interspeech.2011-169","article-title":"Conversational speech transcription using context-dependent deep neural networks","author":"seide","year":"2011","journal-title":"InterSpeech"},{"key":"ref6","author":"bregman","year":"1994","journal-title":"Auditory Scene Analysis The Perceptual Organization of Sound"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/TSA.2005.858005"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1121\/1.1907229"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2015.2444659"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2009.02.006"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2011.2134090"},{"key":"ref9","volume":"7","author":"cooke","year":"2005","journal-title":"Modelling Auditory Processing and Organisation"},{"key":"ref1","article-title":"Roles of pre-training and fine-tuning in context-dependent dbn-hmms for real-world speech recognition","author":"yu","year":"2010","journal-title":"NIPS 2010 Workshop on Deep Learning and Unsupervised Feature Learning"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2013.2291240"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2015.2468583"},{"key":"ref21","doi-asserted-by":"crossref","first-page":"91","DOI":"10.1007\/978-3-319-22482-4_11","article-title":"Speech enhancement with lstm recurrent neural networks and its application to noise-robust asr","author":"weninger","year":"2015","journal-title":"Latent Variable Analysis and Signal Separation"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-1176"},{"key":"ref23","author":"hershey","year":"2015","journal-title":"Deep clustering Discriminative embeddings for segmentation and separation"},{"key":"ref26","author":"garofolo","year":"1993","journal-title":"CSR-I (WSJ0) Complete LDC93S6A"},{"key":"ref25","article-title":"Tutorial: Supervised speech separation","author":"wang","year":"2016","journal-title":"ICASSP"}],"event":{"name":"2017 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","location":"New Orleans, LA","start":{"date-parts":[[2017,3,5]]},"end":{"date-parts":[[2017,3,9]]}},"container-title":["2017 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/7943262\/7951776\/07952154.pdf?arnumber=7952154","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T20:58:19Z","timestamp":1750366699000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/7952154\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017,3]]},"references-count":29,"URL":"https:\/\/doi.org\/10.1109\/icassp.2017.7952154","relation":{},"subject":[],"published":{"date-parts":[[2017,3]]}}}