{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T21:13:52Z","timestamp":1740172432897,"version":"3.37.3"},"reference-count":54,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62176182","62072335","62201314"],"award-info":[{"award-number":["62176182","62072335","62201314"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100007129","name":"Natural Science Foundation of Shandong Province","doi-asserted-by":"publisher","award":["ZR2020QF007"],"award-info":[{"award-number":["ZR2020QF007"]}],"id":[{"id":"10.13039\/501100007129","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Grant-in-Aid for Scientific Research","award":["21H03463"],"award-info":[{"award-number":["21H03463"]}]},{"name":"Fund for the Promotion of Joint International Research","award":["20KK0233"],"award-info":[{"award-number":["20KK0233"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2023]]},"DOI":"10.1109\/taslp.2023.3302237","type":"journal-article","created":{"date-parts":[[2023,8,4]],"date-time":"2023-08-04T17:28:23Z","timestamp":1691170103000},"page":"3206-3220","source":"Crossref","is-referenced-by-count":0,"title":["Unsupervised Deep Unfolded Representation Learning for Singing Voice Separation"],"prefix":"10.1109","volume":"31","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1900-8259","authenticated-orcid":false,"given":"Weitao","family":"Yuan","sequence":"first","affiliation":[{"name":"Tianjin Key Laboratory of Autonomous Intelligence Technology and Systems, School of Software, Tiangong University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4713-5012","authenticated-orcid":false,"given":"Shengbei","family":"Wang","sequence":"additional","affiliation":[{"name":"Tianjin Key Laboratory of Autonomous Intelligence Technology and Systems, School of Software, Tiangong University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2685-4437","authenticated-orcid":false,"given":"Jianming","family":"Wang","sequence":"additional","affiliation":[{"name":"Centre for Engineering Faculty, Tiangong University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6605-2052","authenticated-orcid":false,"given":"Masashi","family":"Unoki","sequence":"additional","affiliation":[{"name":"School of Information Science, Japan Advanced Institute of Science and Technology, Nomi, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8393-5703","authenticated-orcid":false,"given":"Wenwu","family":"Wang","sequence":"additional","affiliation":[{"name":"Centre for Vision, Speech and Signal Processing, University of Surrey, Guildford, U.K."}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"doi-asserted-by":"publisher","key":"ref13","DOI":"10.21105\/joss.01667"},{"doi-asserted-by":"publisher","key":"ref12","DOI":"10.1109\/TASLP.2015.2396681"},{"doi-asserted-by":"publisher","key":"ref15","DOI":"10.1109\/ICASSP49357.2023.10096956"},{"doi-asserted-by":"publisher","key":"ref14","DOI":"10.21105\/joss.02154"},{"key":"ref53","first-page":"1","article-title":"ReduNet: A white-box deep network from the principle of maximizing rate reduction","volume":"23","author":"chan","year":"2022","journal-title":"J Mach Learn Res"},{"key":"ref52","first-page":"9422","article-title":"Learning diverse and discriminative representations via the principle of maximal coding rate reduction","author":"yu","year":"0","journal-title":"Proc Adv Neural Inf Process Sys"},{"doi-asserted-by":"publisher","key":"ref11","DOI":"10.1109\/TASL.2013.2266773"},{"doi-asserted-by":"publisher","key":"ref10","DOI":"10.1109\/ICASSP.2012.6287816"},{"doi-asserted-by":"publisher","key":"ref54","DOI":"10.1109\/TPAMI.2007.1085"},{"year":"2020","author":"mimilakis","article-title":"Revisiting representation learning for singing voice separation with sinkhorn distances","key":"ref17"},{"doi-asserted-by":"publisher","key":"ref16","DOI":"10.23919\/Eusipco47968.2020.9287352"},{"doi-asserted-by":"publisher","key":"ref19","DOI":"10.1109\/ICASSP.2018.8461822"},{"doi-asserted-by":"publisher","key":"ref18","DOI":"10.1109\/IJCNN.2018.8489565"},{"doi-asserted-by":"publisher","key":"ref51","DOI":"10.1109\/ICASSP40776.2020.9054721"},{"doi-asserted-by":"publisher","key":"ref50","DOI":"10.1109\/ICASSP.2019.8683855"},{"doi-asserted-by":"publisher","key":"ref46","DOI":"10.1007\/s10851-014-0523-2"},{"key":"ref45","first-page":"2292","article-title":"Sinkhorn distances: Lightspeed computation of optimal transport","author":"cuturi","year":"0","journal-title":"Proc Adv Neural Inf Process Sys"},{"year":"2017","author":"rafii","article-title":"The MUSDB18 corpus for music separation","key":"ref48"},{"doi-asserted-by":"publisher","key":"ref47","DOI":"10.1137\/15M1032600"},{"key":"ref42","article-title":"The singular values of convolutional layers","author":"sedghi","year":"0","journal-title":"Proc 7th Int Conf Learn Represent"},{"doi-asserted-by":"publisher","key":"ref41","DOI":"10.1007\/s00371-020-02029-7"},{"doi-asserted-by":"publisher","key":"ref44","DOI":"10.1137\/17M1140431"},{"doi-asserted-by":"publisher","key":"ref43","DOI":"10.1007\/978-1-4419-9467-7"},{"doi-asserted-by":"publisher","key":"ref49","DOI":"10.1109\/WASPAA.2019.8937253"},{"key":"ref8","first-page":"192","article-title":"Investigating U-nets with various intermediate blocks for spectrogram-based singing voice separation","author":"choi","year":"0","journal-title":"Proc 21th Int Soc Music Inf Ret Conf"},{"doi-asserted-by":"publisher","key":"ref7","DOI":"10.1109\/ICASSP39728.2021.9413723"},{"key":"ref9","doi-asserted-by":"crossref","first-page":"1361","DOI":"10.1109\/TASL.2009.2020886","article-title":"Monaural musical sound separation based on pitch and common amplitude modulation","volume":"17","author":"li","year":"2009","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"doi-asserted-by":"publisher","key":"ref4","DOI":"10.1109\/TASLP.2021.3091817"},{"doi-asserted-by":"publisher","key":"ref3","DOI":"10.1109\/TASLP.2019.2952013"},{"doi-asserted-by":"publisher","key":"ref6","DOI":"10.1109\/ICASSP43922.2022.9747612"},{"doi-asserted-by":"publisher","key":"ref5","DOI":"10.1109\/TASLP.2022.3140561"},{"key":"ref40","first-page":"703","article-title":"Optimal spectral transportation with application to music transcription","author":"flamary","year":"0","journal-title":"Proc Adv Neural Inf Process Sys"},{"doi-asserted-by":"publisher","key":"ref35","DOI":"10.1109\/TNNLS.2020.3005348"},{"key":"ref34","first-page":"83:1","article-title":"Convolutional neural networks analyzed via convolutional sparse coding","volume":"18","author":"papyan","year":"2017","journal-title":"J Mach Learn Res"},{"doi-asserted-by":"publisher","key":"ref37","DOI":"10.1561\/2200000073"},{"doi-asserted-by":"publisher","key":"ref36","DOI":"10.1109\/TASL.2006.885253"},{"doi-asserted-by":"publisher","key":"ref31","DOI":"10.1093\/bioinformatics\/btz094"},{"key":"ref30","first-page":"2377","article-title":"Training very deep networks","author":"srivastava","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"doi-asserted-by":"publisher","key":"ref33","DOI":"10.1109\/ICASSP.2017.7952123"},{"key":"ref32","first-page":"3371","article-title":"Stacked denoising autoencoders: Learning useful representations in a deep network with a local denoising criterion","volume":"11","author":"vincent","year":"2010","journal-title":"J Mach Learn Res"},{"doi-asserted-by":"publisher","key":"ref2","DOI":"10.1109\/TASLP.2018.2825440"},{"doi-asserted-by":"publisher","key":"ref1","DOI":"10.1109\/TASL.2006.889789"},{"key":"ref39","article-title":"Large scale optimal transport and mapping estimation","author":"seguy","year":"0","journal-title":"Proc 6th Int Conf Learn Represent"},{"key":"ref38","first-page":"957","article-title":"From word embeddings to document distances","author":"kusner","year":"0","journal-title":"Proc 32nd Int Conf Mach Learn"},{"doi-asserted-by":"publisher","key":"ref24","DOI":"10.1109\/MSP.2020.3016905"},{"year":"2009","author":"mallat","journal-title":"A Wavelet Tour of Signal Processing - The Sparse Way","key":"ref23"},{"doi-asserted-by":"publisher","key":"ref26","DOI":"10.1109\/TPAMI.2013.50"},{"doi-asserted-by":"publisher","key":"ref25","DOI":"10.1109\/ICASSP40776.2020.9054172"},{"key":"ref20","first-page":"332","article-title":"Reducing interference with phase recovery in NN-based monaural singing voice separation","author":"magron","year":"0","journal-title":"Proc Annu Conf Int Speech Commun Assoc"},{"doi-asserted-by":"publisher","key":"ref22","DOI":"10.1109\/ICASSP40776.2020.9053513"},{"year":"2019","author":"d\u00e9fossez","article-title":"Music source separation in the waveform domain","key":"ref21"},{"key":"ref28","first-page":"5998","article-title":"Attention is all you need","author":"vaswani","year":"0","journal-title":"Proc Adv Neural Inf Process Sys"},{"doi-asserted-by":"publisher","key":"ref27","DOI":"10.1109\/JPROC.2023.3247480"},{"key":"ref29","article-title":"A trainable optimal transport embedding for feature aggregation and its relationship to attention","author":"mialon","year":"0","journal-title":"Proc 9th Int Conf Learn Represent"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/9970249\/10209210.pdf?arnumber=10209210","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,9,18]],"date-time":"2023-09-18T18:07:47Z","timestamp":1695060467000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10209210\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"references-count":54,"URL":"https:\/\/doi.org\/10.1109\/taslp.2023.3302237","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"type":"print","value":"2329-9290"},{"type":"electronic","value":"2329-9304"}],"subject":[],"published":{"date-parts":[[2023]]}}}