{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,8]],"date-time":"2025-11-08T13:05:16Z","timestamp":1762607116649,"version":"3.28.0"},"reference-count":29,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017,9]]},"DOI":"10.1109\/mlsp.2017.8168166","type":"proceedings-article","created":{"date-parts":[[2017,12,13]],"date-time":"2017-12-13T19:22:14Z","timestamp":1513192934000},"page":"1-6","source":"Crossref","is-referenced-by-count":8,"title":["Learning embeddings for speaker clustering based on voice equality"],"prefix":"10.1109","author":[{"given":"Yanick X.","family":"Lukic","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Carlo","family":"Vogt","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Oliver","family":"Durr","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Thilo","family":"Stadelmann","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2014.09.003"},{"key":"ref11","first-page":"1","article-title":"Going deeper with convolutions","author":"liu","year":"2015","journal-title":"Proc IEEE CVPR"},{"key":"ref12","first-page":"1097","article-title":"Imagenet classification with deep convolutional neural networks","author":"sutskever","year":"2012","journal-title":"Adv NIPS"},{"key":"ref13","first-page":"1995","article-title":"Convolutional networks for images, speech, and time series","volume":"3361","year":"1995","journal-title":"The Handbook of Brain Theory and Neural Networks"},{"key":"ref14","article-title":"Wavenet: A generative model for raw audio","volume":"abs 1609 3499","author":"den oord","year":"2016","journal-title":"CoRR"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2017.2657381"},{"key":"ref16","first-page":"2185","article-title":"Dnn-based speaker clustering for speaker diarisation","year":"2016","journal-title":"Proc INTERSPEECH"},{"key":"ref17","first-page":"686","article-title":"Application of convolutional neural networks to speaker recognition in noisy conditions","author":"lei","year":"2014","journal-title":"Proc INTERSPEECH"},{"key":"ref18","first-page":"3096","article-title":"Speaker diarization with i-vectors from dnn senone posteriors","author":"garcia-romero","year":"2015","journal-title":"Proc INTERSPEECH"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.2011.2167240"},{"key":"ref28","article-title":"Adam: A method for stochastic optimization","volume":"abs 1412 6980","year":"2014","journal-title":"CoRR"},{"key":"ref4","first-page":"1096","article-title":"Unsupervised feature learning for audio classification using convolutional deep belief networks","author":"pham","year":"2009","journal-title":"Adv NIPS"},{"key":"ref27","first-page":"2579","article-title":"Visualizing data using t-sne","volume":"9","author":"der maaten","year":"2008","journal-title":"JMLR"},{"key":"ref3","first-page":"185","article-title":"Unfolding speaker clustering potential: a biomimetic approach","year":"2009","journal-title":"Proc ACM Multimedia"},{"key":"ref6","doi-asserted-by":"crossref","first-page":"1798","DOI":"10.1109\/TPAMI.2013.50","article-title":"Representation learning: A review and new perspectives","volume":"35","author":"courville","year":"2013","journal-title":"IEEE Trans PAMI"},{"key":"ref29","article-title":"Adadelta: An adaptive learning rate method","volume":"abs 1212 5701","year":"2012","journal-title":"CoRR"},{"key":"ref5","doi-asserted-by":"crossref","first-page":"91","DOI":"10.1016\/0167-6393(95)00009-D","article-title":"Speaker identification and verification using gaussian mixture speaker models","volume":"17","year":"1995","journal-title":"Speech Communication"},{"journal-title":"Automatic Speech Recognition","year":"2012","key":"ref8"},{"key":"ref7","doi-asserted-by":"crossref","first-page":"436","DOI":"10.1038\/nature14539","article-title":"Deep learning","volume":"521","author":"bengio","year":"2015","journal-title":"Nature"},{"key":"ref2","first-page":"356","article-title":"Speaker diarization: A review of recent research","volume":"20","author":"bozonnet","year":"2012","journal-title":"IEEE Trans ASLP"},{"key":"ref9","first-page":"1","article-title":"Stadelmann, &#x201C;Speaker identification and clustering using convolutional neural networks","author":"vogt d\u00fcrr","year":"2016","journal-title":"Proc IEEE MLSP"},{"journal-title":"Fundamentals of Speaker Recognition","year":"2011","key":"ref1"},{"key":"ref20","first-page":"402","article-title":"Artificial neural network features for speaker diarization","author":"stolcke","year":"2014","journal-title":"IEEE SLT Workshop"},{"key":"ref22","article-title":"Neural network-based clustering using pairwise constraints","volume":"abs 1511 6321","year":"2015","journal-title":"CoRR"},{"key":"ref21","first-page":"2082","article-title":"Speaker diarization through speaker embeddings","author":"bousquet","year":"2015","journal-title":"Proc EUSIPCO"},{"key":"ref24","first-page":"298","article-title":"Extracting speaker-specific information with a regularized siamese deep network","year":"2011","journal-title":"Adv NIPS"},{"key":"ref23","article-title":"Deep image category discovery using a transferred similarity function","volume":"abs 1612 1253","author":"lv","year":"2016","journal-title":"CoRR"},{"key":"ref26","article-title":"Batch normalization: Accelerating deep network training by reducing internal covariate shift","volume":"abs 1502 3167","year":"2015","journal-title":"CoRR"},{"key":"ref25","doi-asserted-by":"crossref","first-page":"1091","DOI":"10.1016\/j.sigpro.2007.11.017","article-title":"Speaker segmentation and clustering","volume":"88","author":"moschou","year":"2008","journal-title":"Signal Processing"}],"event":{"name":"2017 IEEE 27th International Workshop on Machine Learning for Signal Processing (MLSP)","start":{"date-parts":[[2017,9,25]]},"location":"Tokyo","end":{"date-parts":[[2017,9,28]]}},"container-title":["2017 IEEE 27th International Workshop on Machine Learning for Signal Processing (MLSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8122073\/8168099\/08168166.pdf?arnumber=8168166","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,10,8]],"date-time":"2019-10-08T00:16:01Z","timestamp":1570493761000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/8168166\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017,9]]},"references-count":29,"URL":"https:\/\/doi.org\/10.1109\/mlsp.2017.8168166","relation":{},"subject":[],"published":{"date-parts":[[2017,9]]}}}