{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,18]],"date-time":"2026-02-18T00:31:20Z","timestamp":1771374680234,"version":"3.50.1"},"reference-count":78,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2021]]},"DOI":"10.1109\/taslp.2021.3098764","type":"journal-article","created":{"date-parts":[[2021,7,21]],"date-time":"2021-07-21T20:41:10Z","timestamp":1626900070000},"page":"2575-2590","source":"Crossref","is-referenced-by-count":10,"title":["Guided Generative Adversarial Neural Network for Representation Learning and Audio Generation Using Fewer Labelled Audio Data"],"prefix":"10.1109","volume":"29","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8882-5194","authenticated-orcid":false,"given":"Kazi Nazmul","family":"Haque","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rajib","family":"Rana","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5301-8314","authenticated-orcid":false,"given":"Jiajun","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1382-9929","authenticated-orcid":false,"given":"John","family":"Hansen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nicholas","family":"Cummins","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4075-4072","authenticated-orcid":false,"given":"Carlos","family":"Busso","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6478-8699","authenticated-orcid":false,"given":"Bjorn","family":"Schuller","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1016\/0047-259X(82)90077-X"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2011.5946971"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01216-8_14"},{"key":"ref76","first-page":"2579","article-title":"Visualizing data using t-sne","volume":"9","author":"maaten","year":"2008","journal-title":"J Mach Learn Res"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2636"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00198"},{"key":"ref75","article-title":"Mixed precision training","author":"micikevicius","year":"2017"},{"key":"ref38","first-page":"1679","article-title":"Self-supervised generalisation with meta auxiliary learning","author":"liu","year":"2019","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-63"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3030489"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2922832"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2957889"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2016.2530401"},{"key":"ref37","first-page":"649","article-title":"Colorful image colorization","volume":"9907","author":"zhang","year":"2016","journal-title":"Proc Eur Conf Comput Vis"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2938863"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3065234"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2690564"},{"key":"ref60","article-title":"Semi-supervised conditional Gans","author":"sricharan","year":"2017"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00152"},{"key":"ref61","article-title":"Adversarial feature learning","author":"donahue","year":"2016"},{"key":"ref63","first-page":"2172","article-title":"Infogan: Interpretable representation learning by information maximizing generative adversarial nets","volume":"29","author":"chen","year":"2016","journal-title":"Adv Neural Inf Process Syst"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2236"},{"key":"ref64","article-title":"Speech commands: A dataset for limited-vocabulary speech recognition","author":"warden","year":"2018"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.23919\/EUSIPCO.2018.8553236"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"ref29","first-page":"419","article-title":"The large time-frequency analysis toolbox 2.0","author":"pr?\u0161a","year":"2013","journal-title":"Proc Int Symp Comput Music Multidisciplinary Res"},{"key":"ref66","first-page":"1068","article-title":"Neural audio synthesis of musical notes with wavenet autoencoders","author":"engel","year":"2017","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref67","first-page":"2234","article-title":"Improved techniques for training gans","volume":"29","author":"salimans","year":"2016","journal-title":"Adv Neural Inf Process Syst"},{"key":"ref68","first-page":"6626","article-title":"Gans trained by a two time-scale update rule converge to a local nash equilibrium","volume":"30","author":"heusel","year":"2017","journal-title":"Adv Neural Inf Process Syst"},{"key":"ref69","article-title":"A note on the inception score","author":"barratt","year":"2018"},{"key":"ref2","article-title":"Progressive growing of GANs for improved quality, stability, and variation","author":"karras","year":"2018","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref1","first-page":"2672","article-title":"Generative adversarial nets","author":"goodfellow","year":"2014","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1984.1164317"},{"key":"ref21","article-title":"Char2wav: End-to-End speech synthesis","author":"sotelo","year":"2017","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2746264"},{"key":"ref23","article-title":"Samplernn: An unconditional end-to-end neural audio generation model","author":"mehri","year":"2017","journal-title":"Proc Int Conf Learn Representations (ICLR"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2018.2798811"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.2970241"},{"key":"ref50","first-page":"17","article-title":"Sequence to sequence autoencoders for unsupervised representation learning from audio","author":"amiriparian","year":"2017","journal-title":"Proc DCASE 2017 Workshop"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2690563"},{"key":"ref59","article-title":"Unsupervised and semi-supervised learning with categorical generative adversarial networks","author":"springenberg","year":"2015"},{"key":"ref58","article-title":"Learning multiple layers of features from tiny images","author":"krizhevsky","year":"2012"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.425"},{"key":"ref56","article-title":"Reading digits in natural images with unsupervised feature learning","author":"netzer","year":"2011","journal-title":"Proc NIPS Workshop Deep Learn Unsupervised Feature Learn"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-71249-9_8"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-883"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952656"},{"key":"ref52","article-title":"Auto-encoding variational bayes","author":"kingma","year":"2013"},{"key":"ref10","first-page":"4352","article-title":"Adversarial generation of time-frequency features with application in audio synthesis","author":"marafioti","year":"2019","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01061"},{"key":"ref11","article-title":"Unsupervised representation learning with deep convolutional generative adversarial networks","author":"radford","year":"2016"},{"key":"ref12","first-page":"4183","article-title":"High-fidelity image generation with fewer labels","author":"lucic","year":"2019","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref13","first-page":"4114","article-title":"Challenging common assumptions in the unsupervised learning of disentangled representations","author":"locatello","year":"2019","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref14","first-page":"14910","article-title":"Melgan: Generative adversarial networks for conditional waveform synthesis","author":"kumar","year":"2019","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383551"},{"key":"ref16","article-title":"WaveNet: A generative model for raw audio","author":"van den oord","year":"2016","journal-title":"Proc IEEE Workshop Speech Synth"},{"key":"ref17","first-page":"3918","article-title":"Parallel wavenet: Fast high-fidelity speech synthesis","author":"oord","year":"2018","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref18","article-title":"Clarinet: Parallel wave generation in end-to-end text-to-speech","author":"ping","year":"2019","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"ref4","article-title":"Large scale GAN training for high fidelity natural image synthesis","author":"brock","year":"2019","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00453"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053795"},{"key":"ref5","first-page":"10541","article-title":"Large scale adversarial representation learning","author":"donahue","year":"2019","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref8","article-title":"Adversarial audio synthesis","author":"donahue","year":"2019","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref7","article-title":"High fidelity speech synthesis with adversarial networks","author":"bi?kowski","year":"2020","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref49","article-title":"Deep representation learning in speech processing: Challenges, recent advances, and future trends","author":"latif","year":"2020"},{"key":"ref9","article-title":"GANSynth: Dversarial neural audio synthesis","author":"engel","year":"2019","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.findings-emnlp.106"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1873"},{"key":"ref48","article-title":"Effectiveness of self-supervised pre-training for speech recognition","author":"baevski","year":"2019"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054548"},{"key":"ref42","article-title":"Representation learning with contrastive predictive coding","author":"oord","year":"2018"},{"key":"ref41","article-title":"Unsupervised Representation Learning by Predicting Image Rotations","author":"gidaris","year":"2018","journal-title":"CoRR"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054057"},{"key":"ref43","article-title":"Learning audio representations via phase prediction","author":"quitry","year":"2019"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/9289074\/09492807.pdf?arnumber=9492807","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,5,10]],"date-time":"2022-05-10T14:53:57Z","timestamp":1652194437000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9492807\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021]]},"references-count":78,"URL":"https:\/\/doi.org\/10.1109\/taslp.2021.3098764","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"value":"2329-9290","type":"print"},{"value":"2329-9304","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021]]}}}