{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,4,26]],"date-time":"2025-04-26T05:28:10Z","timestamp":1745645290666,"version":"3.37.3"},"reference-count":61,"publisher":"IEEE","funder":[{"DOI":"10.13039\/501100001321","name":"National Research Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001321","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,1,19]]},"DOI":"10.1109\/slt48900.2021.9383594","type":"proceedings-article","created":{"date-parts":[[2021,3,25]],"date-time":"2021-03-25T20:46:54Z","timestamp":1616705214000},"page":"919-926","source":"Crossref","is-referenced-by-count":11,"title":["Acoustic Word Embeddings for Zero-Resource Languages Using Self-Supervised Contrastive Learning and Multilingual Adaptation"],"prefix":"10.1109","author":[{"given":"Christiaan","family":"Jacobs","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yevgen","family":"Matusevych","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Herman","family":"Kamper","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"article-title":"Learning problem-agnostic speech repre-sentations from multiple self-supervised tasks","year":"2019","author":"pascual","key":"ref39"},{"article-title":"Unsupervised representation learning by predicting image rotations","year":"2018","author":"gidaris","key":"ref38"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2016.7846310"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2828"},{"article-title":"Improved acoustic word embeddings for zero-resource lan-guages using multilingual transfer","year":"2020","author":"kamper","key":"ref31"},{"key":"ref30","article-title":"Multi-lingual acoustic word embedding models for processing zero-resource languages","author":"kamper","year":"2020","journal-title":"Proc ICASSP"},{"key":"ref37","article-title":"Unsupervised learning of visual representations by solving jigsaw puzzles","author":"noroozi","year":"2016","journal-title":"Proc ECCV"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.167"},{"key":"ref35","article-title":"A critical analysis of self-supervision, or what we can learn from a single image","author":"asano","year":"2020","journal-title":"Proc ICLR"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.226"},{"key":"ref60","article-title":"Analyzing autoencoder-based acoustic word embeddings","author":"matusevych","year":"2020","journal-title":"BAICS Workshop ICLR"},{"key":"ref61","first-page":"2579","article-title":"Viualizing data using t-SNE","volume":"9","author":"van der maaten","year":"2008","journal-title":"J Mach Learn Res"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683639"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-82"},{"article-title":"Acoustic word embedding system for code-switching query-by-example spoken term detection","year":"2020","author":"ma","key":"ref29"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2904"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639245"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-2364"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682553"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-2341"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683868"},{"article-title":"Contextual joint factor acoustic embeddings","year":"2019","author":"shi","key":"ref23"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9003929"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682903"},{"key":"ref50","first-page":"207","article-title":"Distance metric learning for large margin nearest neighbor classification","volume":"10","author":"weinberger","year":"2009","journal-title":"J Mach Learn Res"},{"key":"ref51","first-page":"1109","article-title":"Large scale online learning of image similarity through ranking","volume":"11","author":"chechik","year":"2010","journal-title":"J Mach Learn Res"},{"key":"ref59","doi-asserted-by":"crossref","DOI":"10.21437\/Interspeech.2011-304","article-title":"Rapid evaluation of speech representations for spoken term discovery","author":"carlin","year":"2011","journal-title":"Proc INTERSPEECH"},{"key":"ref58","article-title":"Adam: A method for stochastic optimization","author":"kingma","year":"2015","journal-title":"Proc ICLR"},{"key":"ref57","article-title":"Multilingual and unsupervised subword modeling for zero resource languages","author":"hermann","year":"2020","journal-title":"Comput Speech Language"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639248"},{"key":"ref55","article-title":"Neural transfer learning for natural language processing","author":"ruder","year":"2019","journal-title":"Ph D thesis"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2009.191"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639241"},{"article-title":"In defense of the triplet loss for person re-identification","year":"2017","author":"hermans","key":"ref52"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2017.8269008"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2019.2929415"},{"article-title":"End-to-end ASR: from supervised to semi-supervised learning with modern architectures","year":"2019","author":"synnaeve","key":"ref40"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2398"},{"key":"ref13","article-title":"An un-supervised probability model for speech-to-translation alignment of low-resource languages","author":"anastasopoulos","year":"2016","journal-title":"Proc EMNLP"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2013.6707765"},{"key":"ref15","article-title":"Word embeddings for speech recognition","author":"bengio","year":"2014","journal-title":"Proc INTERSPEECH"},{"key":"ref16","article-title":"Multi-view recurrent neural acoustic word embeddings","author":"he","year":"2017","journal-title":"Proc ICLR"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2017.2759726"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462002"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639553"},{"article-title":"Improved audio embeddings by adjacency-based clustering with applications in spoken term detection","year":"2018","author":"huang","key":"ref4"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7179089"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2007.909282"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1010"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2224"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2011.6163965"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2145"},{"key":"ref9","doi-asserted-by":"crossref","DOI":"10.21437\/Interspeech.2020-1738","article-title":"Unsupervised discovery of recurring speech patterns using probabilistic adaptive metrics","author":"r\u00e4s\u00e4nen","year":"2020"},{"article-title":"A simple framework for contrastive learning of visual representations","year":"2020","author":"chen","key":"ref46"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2362"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472619"},{"key":"ref47","article-title":"Improved deep metric learning with multi-class N-pair loss objective","author":"sohn","year":"2016","journal-title":"Proc NeurIPS"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054224"},{"key":"ref41","article-title":"vq-wav2vec: Self-supervised learning of discrete speech representations","author":"baevski","year":"2020","journal-title":"Proc ICLR"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053569"},{"key":"ref43","article-title":"Unsupervised pre-training of bidirectional speech encoders via masked reconstruction","author":"wang","year":"2020","journal-title":"Proc ICASSP"}],"event":{"name":"2021 IEEE Spoken Language Technology Workshop (SLT)","start":{"date-parts":[[2021,1,19]]},"location":"Shenzhen, China","end":{"date-parts":[[2021,1,22]]}},"container-title":["2021 IEEE Spoken Language Technology Workshop (SLT)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9383468\/9383452\/09383594.pdf?arnumber=9383594","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,22]],"date-time":"2022-12-22T13:16:28Z","timestamp":1671714988000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9383594\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,1,19]]},"references-count":61,"URL":"https:\/\/doi.org\/10.1109\/slt48900.2021.9383594","relation":{},"subject":[],"published":{"date-parts":[[2021,1,19]]}}}