{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,31]],"date-time":"2026-03-31T19:54:45Z","timestamp":1774986885291,"version":"3.50.1"},"reference-count":52,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"6","license":[{"start":{"date-parts":[[2022,10,1]],"date-time":"2022-10-01T00:00:00Z","timestamp":1664582400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2022,10,1]],"date-time":"2022-10-01T00:00:00Z","timestamp":1664582400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,10,1]],"date-time":"2022-10-01T00:00:00Z","timestamp":1664582400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE J. Sel. Top. Signal Process."],"published-print":{"date-parts":[[2022,10]]},"DOI":"10.1109\/jstsp.2022.3202093","type":"journal-article","created":{"date-parts":[[2022,8,26]],"date-time":"2022-08-26T19:40:53Z","timestamp":1661542853000},"page":"1402-1414","source":"Crossref","is-referenced-by-count":12,"title":["Decorrelating Feature Spaces for Learning General-Purpose Audio Representations"],"prefix":"10.1109","volume":"16","author":[{"given":"Sreyan","family":"Ghosh","sequence":"first","affiliation":[{"name":"Speech Lab, Department of Electrical Engineering, Indian Institute of Technology Madras, Chennai, Tamil Nadu, India"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3100-9342","authenticated-orcid":false,"given":"Ashish","family":"Seth","sequence":"additional","affiliation":[{"name":"Speech Lab, Department of Electrical Engineering, Indian Institute of Technology Madras, Chennai, Tamil Nadu, India"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"S","family":"Umesh","sequence":"additional","affiliation":[{"name":"Speech Lab, Department of Electrical Engineering, Indian Institute of Technology Madras, Chennai, Tamil Nadu, India"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.73"},{"key":"ref2","article-title":"Self-labelling via simultaneous clustering and representation learning","author":"Asano","year":"2019"},{"key":"ref3","first-page":"12449","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Baevski","year":"2020"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1007\/s10579-008-9076-6"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01264-9_9"},{"key":"ref6","first-page":"9912","article-title":"Unsupervised learning of visual features by contrasting cluster assignments","volume-title":"Proc. 34th Int. Conf. Neural Inf. Process. Syst.","author":"Caron","year":"2021"},{"key":"ref7","first-page":"1597","article-title":"A simple framework for contrastive learning of visual representations","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Chen","year":"2020"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01549"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054438"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1810.04805"},{"key":"ref11","first-page":"1330","article-title":"Neural audio synthesis of musical notes with wavenet autoencoders","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Engel","year":"2017"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3133208"},{"issue":"1","key":"ref13","first-page":"6340","article-title":"auDeep: Unsupervised learning of representations from audio with deep recurrent neural networks","volume-title":"J. Mach. Learn. Res.","volume":"18","author":"Freitag","year":"2017"},{"key":"ref14","first-page":"1330","article-title":"Audio set: An ontology and human-labeled dataset for audio events","volume-title":"Proc. IEEE Int. Conf. Acoust., Speech Signal Process.","author":"Jort","year":"2017"},{"key":"ref15","article-title":"Deep clustering for general-purpose audio representations","author":"Ghosh","year":"2021"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2021-698"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i10.21315"},{"key":"ref18","first-page":"21271","article-title":"Bootstrap your own latent: A new approach to self-supervised learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Grill","year":"2020"},{"key":"ref19","article-title":"The history of speech recognition to the year 2030","author":"Hannun","year":"2021"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/icassp.2017.7952132"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/icassp.2018.8461684"},{"key":"ref25","first-page":"24063","article-title":"Intermediate layers matter in momentum contrastive self supervised learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Kaku","year":"2021"},{"key":"ref26","article-title":"The NTT DCASE2020 challenge task 6 system: Automated audio captioning with keywords and sentence length estimation","author":"Koizumi","year":"2020"},{"key":"ref27","first-page":"3519","article-title":"Similarity of neural network representations revisited","volume-title":"Proc. 36th Int. Conf. Mach. Learn.","author":"Kornblith","year":"2019"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2018-1568"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/icassp40776.2020.9053176"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3095662"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054458"},{"key":"ref32","article-title":"SGDR: Stochastic gradient descent with warm restarts","author":"Loshchilov","year":"2016"},{"key":"ref33","article-title":"A multi-device dataset for urban acoustic scene classification","author":"Mesaros","year":"2018"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2017-950"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN52387.2021.9534474"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/icassp39728.2021.9413528"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1145\/2647868.2655045"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1242"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1111\/2041-210X.13103"},{"key":"ref41","article-title":"Self-supervised audio representation learning for mobile devices","author":"Tagliasacchi","year":"2019"},{"key":"ref42","first-page":"190","article-title":"Effects of word-frequency based pre- and post- processings for audio captioning","volume-title":"Proc. 5th Workshop Detection Classification Acoust. Scenes Events","author":"Takeuchi","year":"2020"},{"key":"ref43","article-title":"A note on connecting barlow twins with negative-sample-free contrastive learning","author":"Hubert Tsai","year":"2021"},{"key":"ref44","article-title":"Hear 2021: Holistic evaluation of audio representations","author":"Turian","year":"2022"},{"key":"ref45","article-title":"Representation learning with contrastive predictive coding","author":"Oord","year":"2018"},{"key":"ref46","article-title":"Free speech... recognition (linux, windows and mac) - voxforge.org","year":"2014"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746790"},{"key":"ref48","article-title":"Speech commands: A dataset for limited-vocabulary speech recognition","author":"Warden","year":"2018"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1775"},{"key":"ref50","article-title":"Large batch training of convolutional networks","author":"You","year":"2017"},{"key":"ref51","first-page":"12310","article-title":"Barlow twins: Self-supervised learning via redundancy reduction","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","author":"Zbontar","year":"2021"},{"key":"ref52","article-title":"mixup: Beyond empirical risk minimization","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zhang","year":"2018"}],"container-title":["IEEE Journal of Selected Topics in Signal Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/4200690\/9923627\/09868132.pdf?arnumber=9868132","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,1]],"date-time":"2024-02-01T12:44:15Z","timestamp":1706791455000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9868132\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10]]},"references-count":52,"journal-issue":{"issue":"6"},"URL":"https:\/\/doi.org\/10.1109\/jstsp.2022.3202093","relation":{},"ISSN":["1932-4553","1941-0484"],"issn-type":[{"value":"1932-4553","type":"print"},{"value":"1941-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,10]]}}}