{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T14:25:51Z","timestamp":1740147951501,"version":"3.37.3"},"reference-count":104,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"6","license":[{"start":{"date-parts":[[2022,10,1]],"date-time":"2022-10-01T00:00:00Z","timestamp":1664582400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2022,10,1]],"date-time":"2022-10-01T00:00:00Z","timestamp":1664582400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,10,1]],"date-time":"2022-10-01T00:00:00Z","timestamp":1664582400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"National Research Foundation of South Africa","award":["120409"],"award-info":[{"award-number":["120409"]}]},{"name":"Google Africa Ph.D. Scholarship"},{"name":"Romanian Ministry of Education and Research"},{"name":"CNCS-UEFISCDI","award":["PN-III-P1-1.1-PD-2019-0918"],"award-info":[{"award-number":["PN-III-P1-1.1-PD-2019-0918"]}]},{"name":"PNCDI III"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE J. Sel. Top. Signal Process."],"published-print":{"date-parts":[[2022,10]]},"DOI":"10.1109\/jstsp.2022.3180220","type":"journal-article","created":{"date-parts":[[2022,6,3]],"date-time":"2022-06-03T19:46:27Z","timestamp":1654285587000},"page":"1454-1466","source":"Crossref","is-referenced-by-count":6,"title":["Keyword Localisation in Untranscribed Speech Using Visually Grounded Speech Models"],"prefix":"10.1109","volume":"16","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2622-7305","authenticated-orcid":false,"given":"Kayode","family":"Olaleye","sequence":"first","affiliation":[{"name":"Stellenbosch University, Stellenbosch, South Africa"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4354-4393","authenticated-orcid":false,"given":"Dan","family":"Oneata","sequence":"additional","affiliation":[{"name":"University Politehnica of Bucharest, Bucharest, Romania"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2980-3475","authenticated-orcid":false,"given":"Herman","family":"Kamper","sequence":"additional","affiliation":[{"name":"Stellenbosch University, Stellenbosch, South Africa"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N16-1109"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-968"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1592"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-503"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2010.5700813"},{"key":"ref6","article-title":"Learning words from images and speech","volume-title":"Proc. Neural Inf. Process. Syst. Workshop Learn","author":"Synnaeve","year":"2014"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2015.7404800"},{"key":"ref8","first-page":"1858","article-title":"Unsupervised learning of spoken language with visual context","volume-title":"Proc. Neural Inf. Process. Syst.","author":"Harwath","year":"2016"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-1047"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01231-1_40"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462396"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683587"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682666"},{"key":"ref14","article-title":"Learning hierarchical discrete linguistic units from visually-grounded speech","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Harwath","year":"2019"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1016\/0022-0965(83)90085-1"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.5860\/choice.31-6014"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.2307\/1131427"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2003.811618"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/COGINF.2007.4341909"},{"key":"ref20","first-page":"1309","article-title":"From phonemes to images: Levels of representation in a recurrent neural model of visually-grounded language learning","volume-title":"Proc. COLING 26th Int. Conf. Comput. Linguistics: Tech. Papers","author":"Gelderloos","year":"2016"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1523"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2012.09.006"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-87"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1162\/089976698300017368"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461761"},{"key":"ref26","first-page":"55","article-title":"Research methods in language documentation","volume-title":"Lang. Documentation Description","volume":"7","author":"Lpke","year":"2010"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1162\/coli_a_00387"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.2973896"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3053391"},{"key":"ref30","first-page":"25","article-title":"Self-supervised multimodal versatile networks","volume-title":"Proc. Neural Inf. Process. Syst.","author":"Alayrac","year":"2020"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1312"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00791"},{"key":"ref33","article-title":"Self-supervised learning from a multi-view perspective","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Tsai","year":"2021"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/p17-1057"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683069"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/K19-1032"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2017-502"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2018.2872106"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-3051"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr.2009.5206848"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2010.5539970"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-016-0981-7"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-020-01316-z"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr.2016.90"},{"key":"ref46","first-page":"6105","article-title":"EfficientNet: Rethinking model scaling for convolutional neural networks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Tan","year":"2019"},{"key":"ref47","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Dosovitskiy","year":"2021"},{"key":"ref48","first-page":"24261","article-title":"MLP-Mixer: An all-MLP architecture for vision","volume-title":"Proc. Neural Inf. Process. Syst.","author":"Tolstikhin","year":"2021"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.coling-main.303"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298668"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.74"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2015-647"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952190"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639622"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461375"},{"key":"ref56","article-title":"Towards localisation of keywords in speech using weak supervision","volume-title":"Proc. SAS Neural Inf. Process. Syst. Workshop","author":"Olaleye","year":"2020"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-435"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1613\/jair.1.12967"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747103"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.21437\/GLU.2017-12"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/ISCAS51556.2021.9401692"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.21437\/SLTU.2018-52"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3120644"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2010.5540112"},{"key":"ref65","first-page":"2764","article-title":"Wsabie: Scaling up to large vocabulary image annotation","volume-title":"Proc. Int. Joint Conf. Artif. Intell.","author":"Weston","year":"2011"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-15561-1_2"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2011.5995466"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00166"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1613\/jair.4900"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1238"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v31i1.11197"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00754"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1437"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00268"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-016-0965-7"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58558-7_38"},{"article-title":"Distilling the knowledge in a neural network","year":"2015","author":"Hinton","key":"ref78"},{"key":"ref79","first-page":"2654","article-title":"Do deep nets really need to be deep?","volume-title":"Proc. Neural Inf. Process. Syst.","author":"Ba","year":"2013"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-021-01453-z"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.309"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00768"},{"key":"ref83","first-page":"892","article-title":"SoundNet: Learning sound representations from unlabeled video","volume-title":"Proc. Neural Inf. Process. Syst.","author":"Aytar","year":"2016"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1145\/3240508.3240578"},{"article-title":"Learning visual models from paired audio-visual examples","year":"2016","author":"Owens","key":"ref85"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-018-1083-5"},{"key":"ref87","first-page":"9758","article-title":"Self-supervised learning by cross-modal audio-video clustering","volume-title":"Proc. Neural Inf. Process. Syst.","author":"Alwassel","year":"2020"},{"issue":"2","key":"ref88","first-page":"137","article-title":"Robust real-time object detection","volume-title":"Int. J. Comput. Vis.","volume":"57","author":"Viola","year":"2001"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10590-1_53"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683275"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.319"},{"article-title":"Neural turing machines","year":"2014","author":"Graves","key":"ref92"},{"key":"ref93","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/d15-1166"},{"key":"ref94","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054678"},{"key":"ref95","first-page":"2127","article-title":"Attention-based deep multiple instance learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Ilse","year":"2018"},{"key":"ref96","first-page":"139","article-title":"Collecting image annotations using amazons mechanical turk","volume-title":"Proc. Workshop Creating Speech Lang. Datah Amazons Mech. Turk","author":"Rashtchian","year":"2010"},{"key":"ref97","article-title":"Very deep convolutional networks for large-scale image recognition","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Simonyan","year":"2015"},{"key":"ref98","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2019-2680"},{"key":"ref99","article-title":"ADAM: A method for stochastic optimization","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Kingma","year":"2015"},{"key":"ref100","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.insights-1.11"},{"key":"ref101","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46454-1_49"},{"key":"ref102","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-017-1059-x"},{"key":"ref103","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00886"},{"key":"ref104","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2021.101275"}],"container-title":["IEEE Journal of Selected Topics in Signal Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/4200690\/9923627\/09787772.pdf?arnumber=9787772","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,1]],"date-time":"2024-02-01T01:47:44Z","timestamp":1706752064000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9787772\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10]]},"references-count":104,"journal-issue":{"issue":"6"},"URL":"https:\/\/doi.org\/10.1109\/jstsp.2022.3180220","relation":{},"ISSN":["1932-4553","1941-0484"],"issn-type":[{"type":"print","value":"1932-4553"},{"type":"electronic","value":"1941-0484"}],"subject":[],"published":{"date-parts":[[2022,10]]}}}