{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T09:49:13Z","timestamp":1784454553954,"version":"3.55.0"},"reference-count":61,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"4","license":[{"start":{"date-parts":[[2019,8,1]],"date-time":"2019-08-01T00:00:00Z","timestamp":1564617600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2019,8,1]],"date-time":"2019-08-01T00:00:00Z","timestamp":1564617600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,8,1]],"date-time":"2019-08-01T00:00:00Z","timestamp":1564617600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100014809","name":"Technology Agency of the Czech Republic","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100014809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"NOSICI"},{"name":"Czech National Science Foundation","award":["19-26934X"],"award-info":[{"award-number":["19-26934X"]}]},{"name":"National Programme of Sustainability"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE J. Sel. Top. Signal Process."],"published-print":{"date-parts":[[2019,8]]},"DOI":"10.1109\/jstsp.2019.2922820","type":"journal-article","created":{"date-parts":[[2019,6,13]],"date-time":"2019-06-13T19:51:37Z","timestamp":1560455497000},"page":"800-814","source":"Crossref","is-referenced-by-count":200,"title":["SpeakerBeam: Speaker Aware Neural Network for Target Speaker Extraction in Speech Mixtures"],"prefix":"10.1109","volume":"13","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4438-8580","authenticated-orcid":false,"given":"Katerina","family":"Zmolikova","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5175-7834","authenticated-orcid":false,"given":"Marc","family":"Delcroix","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Keisuke","family":"Kinoshita","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tsubasa","family":"Ochiai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7487-7150","authenticated-orcid":false,"given":"Tomohiro","family":"Nakatani","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lukas","family":"Burget","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8800-0210","authenticated-orcid":false,"given":"Jan","family":"Cernocky","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2011.5947358"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2010.2081790"},{"key":"ref33","article-title":"Voicefilter: Targeted voice separation by speaker-conditioned spectrogram masking","author":"wang","year":"2018"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461778"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462661"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461533"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2011.6163922"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-1249"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178829"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462665"},{"key":"ref60","article-title":"SpeakerBeam","year":"0"},{"key":"ref61","first-page":"2579","article-title":"Visualizing data using t-SNE","volume":"9","author":"maaten","year":"2008","journal-title":"J Mach Learn Res"},{"key":"ref28","first-page":"2655","article-title":"Speaker-aware neural network based beamformer for speaker extraction in speech mixtures","author":"zmolikova","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2726762"},{"key":"ref29","first-page":"8","article-title":"Learning speaker representation for neural network based multichannel speaker extraction","author":"\u017emol\u00edkov\u00e1","year":"0","journal-title":"Proc IEEE Autom Speech Recognit Understanding Workshop"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1631\/FITEE.1700814"},{"key":"ref1","doi-asserted-by":"crossref","DOI":"10.1007\/978-1-4020-6479-1","volume":"615","author":"makino","year":"2007","journal-title":"Blind Speech Separation"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6853591"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2018.2798821"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472692"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2064307"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2014.7078569"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952155"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-1176"},{"key":"ref50","article-title":"Room impulse response generator","author":"habets","year":"2010"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2012.10.004"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683087"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-1176"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/WASPAA.2013.6701894"},{"key":"ref56","first-page":"367","article-title":"mir_eval: A transparent implementation of common MIR metrics","author":"raffel","year":"0","journal-title":"Proc Int Soc for Music Inf Retrieval Conf"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/TSA.2005.858005"},{"key":"ref54","article-title":"The Kaldi speech recognition toolkit","author":"povey","year":"0","journal-title":"Proc IEEE Workshop Autom Speech Recognit Understanding"},{"key":"ref53","first-page":"249","article-title":"Understanding the difficulty of training deep feedforward neural networks","author":"glorot","year":"0","journal-title":"Proc 13th Int Conf Artif Intell Statist"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2015.7404837"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7471631"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952154"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2014.2352935"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-1346"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1570"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1531"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1205"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICOSP.2014.7015050"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2016.2536478"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2016.2558822"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2013.6707705"},{"key":"ref4","article-title":"Prediction-driven computational auditory scene analysis","author":"ellis","year":"1996"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1006\/csla.1994.1016"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2006.885253"},{"key":"ref5","first-page":"556","article-title":"Algorithms for non-negative matrix factorization","author":"lee","year":"0","journal-title":"Adv Neural Inf Process Syst"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2010.938081"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2008.11.001"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1121\/1.382599"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2018.2842159"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952140"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-552"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461639"},{"key":"ref47","article-title":"CSR-I (WSJ0) complete LDC93S6A","author":"garofolo","year":"1993"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2007.898454"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952118"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7471664"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2012.2205285"}],"container-title":["IEEE Journal of Selected Topics in Signal Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/4200690\/8771267\/08736286.pdf?arnumber=8736286","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,13]],"date-time":"2022-07-13T21:08:16Z","timestamp":1657746496000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8736286\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,8]]},"references-count":61,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.1109\/jstsp.2019.2922820","relation":{},"ISSN":["1932-4553","1941-0484"],"issn-type":[{"value":"1932-4553","type":"print"},{"value":"1941-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,8]]}}}