{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,16]],"date-time":"2026-01-16T00:32:01Z","timestamp":1768523521209,"version":"3.49.0"},"reference-count":20,"publisher":"IEEE","license":[{"start":{"date-parts":[[2019,5,1]],"date-time":"2019-05-01T00:00:00Z","timestamp":1556668800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2019,5,1]],"date-time":"2019-05-01T00:00:00Z","timestamp":1556668800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,5,1]],"date-time":"2019-05-01T00:00:00Z","timestamp":1556668800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019,5]]},"DOI":"10.1109\/icassp.2019.8682632","type":"proceedings-article","created":{"date-parts":[[2019,4,17]],"date-time":"2019-04-17T20:01:56Z","timestamp":1555531316000},"page":"4095-4099","source":"Crossref","is-referenced-by-count":36,"title":["Cross Modal Audio Search and Retrieval with Joint Embeddings Based on Text and Audio"],"prefix":"10.1109","author":[{"given":"Benjamin","family":"Elizalde","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuayb","family":"Zarar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bhiksha","family":"Raj","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.541"},{"key":"ref11","first-page":"1735","article-title":"Dimensionality reduction by learning an invariant mapping","author":"hadsell","year":"2006","journal-title":"IEEE Nul"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1145\/3243250.3243268"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2013.225"},{"key":"ref14","first-page":"486","article-title":"Freesound datasets: a platform for the creation of open audio datasets","author":"fonseca","year":"2017","journal-title":"Proceedings of the 18th International Society for Music Information Retrieval Conference (ISMIR 2017)"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1162"},{"key":"ref16","article-title":"DCASE 2017 challenge setup: Tasks, datasets and baseline system","author":"mesaros","year":"2017","journal-title":"Detection and Classification of Acoustic Scenes and Events"},{"key":"ref17","article-title":"A closer look at weak label learning for audio events","author":"shah","year":"2018"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/1873951.1874246"},{"key":"ref19","first-page":"2825","article-title":"Scikitlearn: Machine learning in Python","volume":"12","author":"pedregosa","year":"2011","journal-title":"Journal of Machine Learning Research"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2041384"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461524"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1145\/1460096.1460115"},{"key":"ref5","article-title":"An overview of cross-media retrieval: Concepts, methodologies, benchmarks and challenges","author":"peng","year":"2017","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2016.2587218"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2002.5745561"},{"key":"ref2","article-title":"Nels-never-ending learner of sounds","author":"elizalde","year":"2017","journal-title":"NIPS Machine Learning for Audio (ML4Audio) Workshop"},{"key":"ref1","first-page":"52","article-title":"Trecvid 2014&#x2013;an overview of the goals, tasks, data, evaluation mechanisms and metrics","author":"over","year":"2014","journal-title":"Proceedings of TRECVID"},{"key":"ref9","article-title":"See, hear, and read: Deep aligned representations","author":"aytar","year":"2017"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1145\/1835449.1835560"}],"event":{"name":"ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","location":"Brighton, United Kingdom","start":{"date-parts":[[2019,5,12]]},"end":{"date-parts":[[2019,5,17]]}},"container-title":["ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8671773\/8682151\/08682632.pdf?arnumber=8682632","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,15]],"date-time":"2022-07-15T03:11:59Z","timestamp":1657854719000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8682632\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,5]]},"references-count":20,"URL":"https:\/\/doi.org\/10.1109\/icassp.2019.8682632","relation":{},"subject":[],"published":{"date-parts":[[2019,5]]}}}