{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T15:11:19Z","timestamp":1784301079929,"version":"3.55.0"},"reference-count":80,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"2","license":[{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62306259"],"award-info":[{"award-number":["62306259"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Research Grants Council of the Hong Kong SAR","award":["C5052-23G"],"award-info":[{"award-number":["C5052-23G"]}]},{"name":"Research Grants Council of the Hong Kong SAR","award":["PolyU25216423"],"award-info":[{"award-number":["PolyU25216423"]}]},{"name":"Research Grants Council of the Hong Kong SAR","award":["PolyU15217424"],"award-info":[{"award-number":["PolyU15217424"]}]},{"name":"Research Grants Council of the Hong Kong SAR","award":["PolyU15218622"],"award-info":[{"award-number":["PolyU15218622"]}]},{"name":"Research Grants Council of the Hong Kong SAR","award":["PolyU15215623"],"award-info":[{"award-number":["PolyU15215623"]}]},{"name":"Research Grants Council of the Hong Kong SAR","award":["PolyU15229824"],"award-info":[{"award-number":["PolyU15229824"]}]},{"DOI":"10.13039\/501100004377","name":"The Hong Kong Polytechnic University","doi-asserted-by":"crossref","award":["P0043563"],"award-info":[{"award-number":["P0043563"]}],"id":[{"id":"10.13039\/501100004377","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100004377","name":"The Hong Kong Polytechnic University","doi-asserted-by":"crossref","award":["P0046094"],"award-info":[{"award-number":["P0046094"]}],"id":[{"id":"10.13039\/501100004377","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100004377","name":"The Hong Kong Polytechnic University","doi-asserted-by":"crossref","award":["P0053699"],"award-info":[{"award-number":["P0053699"]}],"id":[{"id":"10.13039\/501100004377","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100004377","name":"The Hong Kong Polytechnic University","doi-asserted-by":"crossref","award":["P0052694"],"award-info":[{"award-number":["P0052694"]}],"id":[{"id":"10.13039\/501100004377","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100004377","name":"The Hong Kong Polytechnic University","doi-asserted-by":"crossref","award":["P0053758"],"award-info":[{"award-number":["P0053758"]}],"id":[{"id":"10.13039\/501100004377","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Cogn. Dev. Syst."],"published-print":{"date-parts":[[2026,4]]},"DOI":"10.1109\/tcds.2025.3598687","type":"journal-article","created":{"date-parts":[[2025,8,13]],"date-time":"2025-08-13T17:36:03Z","timestamp":1755106563000},"page":"361-372","source":"Crossref","is-referenced-by-count":4,"title":["Typing to Listen at the Cocktail Party: Text-Guided Target Speaker Extraction"],"prefix":"10.1109","volume":"18","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-4364-4556","authenticated-orcid":false,"given":"Xiang","family":"Hao","sequence":"first","affiliation":[{"name":"Department of Data Science and Artificial Intelligence, The Hong Kong Polytechnic University, Hung Hom, Hong Kong SAR"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0135-4188","authenticated-orcid":false,"given":"Jibin","family":"Wu","sequence":"additional","affiliation":[{"name":"Department of Data Science and Artificial Intelligence, the Department of Computing, and the Research Center of Data Science and Artificial Intelligence, The Hong Kong Polytechnic University, Hung Hom, Hong Kong SAR"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2449-1436","authenticated-orcid":false,"given":"Jianwei","family":"Yu","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1584-6282","authenticated-orcid":false,"given":"Chenglin","family":"Xu","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6802-2463","authenticated-orcid":false,"given":"Kay Chen","family":"Tan","sequence":"additional","affiliation":[{"name":"Department of Data Science and Artificial Intelligence and the Research Center of Data Science and Artificial Intelligence, The Hong Kong Polytechnic University, Hung Hom, Hong Kong SAR"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"issue":"5","key":"ref1","doi-asserted-by":"crossref","first-page":"975","DOI":"10.1121\/1.1907229","article-title":"Some experiments on the recognition of speech, with one and with two ears","volume":"25","author":"Colin","year":"1953","journal-title":"J. Acoust. Soc. Amer."},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1162\/0899766054322964"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1038\/nature11020"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1038\/nrn3565"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2016.2647702"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2019.2922820"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.2987429"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053426"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383556"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-2284"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2023.3240008"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548397"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746221"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2018.2795749"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10094306"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10094573"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-27733-7_9048-2"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10097210"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-024-4321-9"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.1055"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i21.30570"},{"key":"ref22","article-title":"Emergent abilities of large language models","author":"Wei","year":"2022"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1016\/j.lindif.2023.102274"},{"key":"ref24","article-title":"\u201cLLAMA 2: Open foundation and fine-tuned chat models,\u201d","year":"2023"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1983.1171927"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1121\/1.400725"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.21437\/Eurospeech.2003-406"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/9780470043387"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2006.1661352"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2006.885253"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2007.366322"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2006-23"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2047419"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ICSDA.2013.6709849"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7471631"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952154"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2915167"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2020-1397"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2022.3153258"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-1176"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462507"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2726762"},{"key":"ref43","article-title":"Listen, think, and understand","author":"Gong","year":"2023"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/WASPAA.2017.8170058"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682377"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1186\/s13636-022-00259-2"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/icassp49357.2023.10095889"},{"key":"ref48","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","author":"Radford","year":"2021"},{"key":"ref49","article-title":"Make-an-audio: Text-to-audio generation with prompt-enhanced diffusion models","author":"Huang","year":"2023"},{"key":"ref50","article-title":"AudioGen: Textually guided audio generation","author":"Kreuk","year":"2023"},{"key":"ref51","first-page":"21450","article-title":"AudioLDM: Text-to-audio generation with latent diffusion models","volume-title":"Proc. 40th Int. Conf. Mach. Learn.","author":"Liu","year":"2023"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/taslp.2024.3399607"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/taslp.2024.3419418"},{"key":"ref54","article-title":"Voicebox: Text-guided multilingual universal speech generation at scale","author":"Le","year":"2023"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-11052"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-10894"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2022-10894"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095266"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1810.04805"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/icassp48485.2024.10447265"},{"key":"ref61","article-title":"Noise2Music: Text-conditioned music generation with diffusion models","author":"Huang","year":"2023"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.5120\/ijca2017915495"},{"key":"ref63","first-page":"1","article-title":"Hierarchical musical instrument separation","volume-title":"Proc. Int. Soc. Music Inf. Retrieval Conf.","author":"Manilow","year":"2020"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1369"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-10717"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2022.3221000"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2020-2210"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1177\/1084713808325306"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-33036-5"},{"key":"ref70","article-title":"LoRA: Low-rank adaptation of large language models","author":"Shen","year":"2021"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683634"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683855"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2826"},{"key":"ref75","article-title":"Montreal forced aligner","author":"Chodroff","year":"2023"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2020"},{"key":"ref77","article-title":"\u201cDeepseek-r1: Incentivizing reasoning capability in LLMS via reinforcement learning,\u201d","author":"DeepSeek","year":"2025"},{"key":"ref78","article-title":"\u201cThe LLAMA 3 herd of models,\u201d","year":"2024"},{"key":"ref79","article-title":"\u201cQwen technical report,\u201d","year":"2023"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054683"}],"container-title":["IEEE Transactions on Cognitive and Developmental Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/7274989\/11477821\/11124510.pdf?arnumber=11124510","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,9]],"date-time":"2026-04-09T19:48:30Z","timestamp":1775764110000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11124510\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4]]},"references-count":80,"journal-issue":{"issue":"2"},"URL":"https:\/\/doi.org\/10.1109\/tcds.2025.3598687","relation":{},"ISSN":["2379-8920","2379-8939"],"issn-type":[{"value":"2379-8920","type":"print"},{"value":"2379-8939","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4]]}}}