{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T04:19:29Z","timestamp":1765253969595,"version":"3.37.3"},"reference-count":62,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"10","license":[{"start":{"date-parts":[[2019,10,1]],"date-time":"2019-10-01T00:00:00Z","timestamp":1569888000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2019,10,1]],"date-time":"2019-10-01T00:00:00Z","timestamp":1569888000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,10,1]],"date-time":"2019-10-01T00:00:00Z","timestamp":1569888000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"Office of the Assistant Secretary of Defense for Health Affairs"},{"name":"Military Suicide Research Consortium","award":["W81XWH-10-2-0181"],"award-info":[{"award-number":["W81XWH-10-2-0181"]}]},{"name":"Psychological Health and Traumatic Brain Injury Research Program","award":["W81XWH-15-1-0632"],"award-info":[{"award-number":["W81XWH-15-1-0632"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2019,10]]},"DOI":"10.1109\/taslp.2019.2921890","type":"journal-article","created":{"date-parts":[[2019,6,11]],"date-time":"2019-06-11T00:49:06Z","timestamp":1560214146000},"page":"1577-1589","source":"Crossref","is-referenced-by-count":37,"title":["Neural Predictive Coding Using Convolutional Neural Networks Toward Unsupervised Learning of Speaker Characteristics"],"prefix":"10.1109","volume":"27","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9498-8536","authenticated-orcid":false,"given":"Arindam","family":"Jati","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0790-7161","authenticated-orcid":false,"given":"Panayiotis","family":"Georgiou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953232"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/45.1890"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/MLSP.2016.7738889"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2009.191"},{"article-title":"Deep speaker: An end-to-end neural speaker embedding system","year":"2017","author":"li","key":"ref31"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2016.7846260"},{"article-title":"TIMIT acoustic-phonetic continuous speech corpus","year":"1993","author":"garofolo","key":"ref37"},{"key":"ref36","first-page":"1096","article-title":"Unsupervised feature learning for audio classification using convolutional deep belief networks","author":"lee","year":"0","journal-title":"Proc Advances Neural Inf Process Syst"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.2002.1021888"},{"article-title":"Multilayer bootstrap network for unsupervised speaker recognition","year":"2015","author":"zhang","key":"ref34"},{"key":"ref60","article-title":"The kaldi speech recognition toolkit","author":"povey","year":"0","journal-title":"Proc IEEE Workshop Autom Speech Recognit Understanding"},{"key":"ref62","first-page":"2579","article-title":"Visualizing data using t-SNE","volume":"9","author":"van der maaten","year":"2008","journal-title":"J Mach Learn Res"},{"key":"ref61","first-page":"26","article-title":"Lecture 6.5-RMSProp: Divide the gradient by a running average of its recent magnitude","volume":"4","author":"tieleman","year":"2012","journal-title":"Neural Netw Mach Learning"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.2011.2167240"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2015.7404844"},{"key":"ref29","first-page":"298","article-title":"Extracting speaker-specific information with a regularized siamese deep network","author":"chen","year":"0","journal-title":"Proc Advances Neural Inf Process Syst"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2012.2236291"},{"journal-title":"Spoken Language Processing A Guide to Theory Algorithm and System Development","year":"2001","author":"huang","key":"ref1"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2013.6707705"},{"journal-title":"Deep Learning","year":"2016","author":"goodfellow","key":"ref22"},{"key":"ref21","first-page":"2341","article-title":"I-vector based speaker recognition on short utterances","author":"kanagasundaram","year":"0","journal-title":"Proc 12th Annu Conf Int Speech Commun Assoc"},{"key":"ref24","doi-asserted-by":"crossref","first-page":"504","DOI":"10.1126\/science.1127647","article-title":"Reducing the dimensionality of data with neural networks","volume":"313","author":"hinton","year":"2006","journal-title":"Science"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/EUSIPCO.2015.7362751"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6854363"},{"key":"ref25","first-page":"3661","article-title":"Improvement of distant-talking speaker identification using bottleneck features of DNN","author":"yamada","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref50","first-page":"686","article-title":"Application of convolutional neural networks to speaker recognition in noisy conditions","author":"mclaren","year":"0","journal-title":"Proc Annu Conf Int Speech Commun Assoc"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/MLSP.2016.7738816"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-950"},{"key":"ref58","first-page":"125","article-title":"TED-LIUM: An automatic speech recognition dedicated corpus","author":"rousseau","year":"0","journal-title":"Proc Lang Resources Eval Conf"},{"key":"ref57","first-page":"999","article-title":"Deep neural network embeddings for text-independent speaker verification","author":"snyder","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref56","first-page":"1929","article-title":"Dropout: A simple way to prevent neural networks from overfitting","volume":"15","author":"srivastava","year":"2014","journal-title":"J Mach Learn Res"},{"key":"ref55","first-page":"448","article-title":"Batch normalization: Accelerating deep network training by reducing internal covariate shift","volume":"37","author":"ioffe","year":"2015","journal-title":"Proceedings of the 32nd Intl Conf on Machine Learning"},{"key":"ref54","article-title":"Rectifier nonlinearities improve neural network acoustic models","volume":"30","author":"maas","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref53","first-page":"905","article-title":"Robust CNN-based speech recognition with Gabor filter kernels","author":"chang","year":"0","journal-title":"Proc Annu Conf Int Speech Commun Assoc"},{"key":"ref52","article-title":"CNN architectures for large-scale audio classification","volume":"abs 1609 9430","author":"hershey","year":"2016","journal-title":"CoRR"},{"journal-title":"Fundamentals of speech recognition","year":"1993","author":"rabiner","key":"ref10"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2005.02.018"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1650"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1006\/dspr.1999.0361"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2004.1325916"},{"key":"ref14","first-page":"2691","article-title":"New map estimators for speaker recognition","author":"kenny","year":"0","journal-title":"Proc EUROSPEECH"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2006.881693"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2064307"},{"key":"ref17","first-page":"1559","article-title":"Support vector machines versus fast scoring in the low-dimensional total variability space for speaker verification","author":"dehak","year":"0","journal-title":"Proc 10th Annu Conf Int Speech Commun Assoc"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2014.7078610"},{"key":"ref19","first-page":"2174","article-title":"I-vectors and ILP clustering adapted to cross-show speaker diarization","author":"dupuy","year":"0","journal-title":"Proc Annu Conf Int Speech Commun Assoc"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2011.2125954"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/j.sigpro.2007.11.017"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/5.628714"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1016\/S0031-3203(02)00034-1"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1980.1163420"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2012.2205597"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2015.2462851"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1121\/1.399423"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1145\/3065386"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1142\/9789812797926_0003"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2014.2339736"},{"article-title":"Very deep convolutional networks for large-scale image recognition","year":"2014","author":"simonyan","key":"ref47"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2006.100"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2005.202"},{"key":"ref44","article-title":"Siamese neural networks for one-shot image recognition","volume":"2","author":"koch","year":"0","journal-title":"Proc Int Conf Mach Learn Deep Learn Workshop"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461375"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/8751213\/08733892.pdf?arnumber=8733892","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,13]],"date-time":"2022-07-13T21:13:26Z","timestamp":1657746806000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8733892\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,10]]},"references-count":62,"journal-issue":{"issue":"10"},"URL":"https:\/\/doi.org\/10.1109\/taslp.2019.2921890","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"type":"print","value":"2329-9290"},{"type":"electronic","value":"2329-9304"}],"subject":[],"published":{"date-parts":[[2019,10]]}}}