{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T23:17:13Z","timestamp":1784330233870,"version":"3.55.0"},"reference-count":90,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"ERC","award":["853489 - DEXIM"],"award-info":[{"award-number":["853489 - DEXIM"]}]},{"name":"DFG 2064\/1","award":["390727645"],"award-info":[{"award-number":["390727645"]}]},{"name":"BMBF","award":["FKZ: 01IS18039A"],"award-info":[{"award-number":["FKZ: 01IS18039A"]}]},{"name":"EPSRC DTA Studentship"},{"DOI":"10.13039\/501100000287","name":"Royal Academy of Engineering","doi-asserted-by":"publisher","award":["RF\\201819\\18\\163"],"award-info":[{"award-number":["RF\\201819\\18\\163"]}],"id":[{"id":"10.13039\/501100000287","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100000266","name":"Engineering and Physical Sciences Research Council","doi-asserted-by":"publisher","award":["EP\/T028572\/1 Visual AI"],"award-info":[{"award-number":["EP\/T028572\/1 Visual AI"]}],"id":[{"id":"10.13039\/501100000266","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Multimedia"],"published-print":{"date-parts":[[2023]]},"DOI":"10.1109\/tmm.2022.3149712","type":"journal-article","created":{"date-parts":[[2022,2,8]],"date-time":"2022-02-08T20:36:48Z","timestamp":1644352608000},"page":"2675-2685","source":"Crossref","is-referenced-by-count":55,"title":["Audio Retrieval With Natural Language Queries: A Benchmark Study"],"prefix":"10.1109","volume":"25","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5807-0576","authenticated-orcid":false,"given":"A. Sophia","family":"Koepke","sequence":"first","affiliation":[{"name":"Explainable Machine Learning Group at the University of T&#x00FC;bingen, T&#x00FC;bingen, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Andreea-Maria","family":"Oncescu","sequence":"additional","affiliation":[{"name":"Visual Geometry Group at the University of Oxford, Oxford, U.K."}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jo\u00e3o F.","family":"Henriques","sequence":"additional","affiliation":[{"name":"Visual Geometry Group at the University of Oxford, Oxford, U.K."}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1432-7747","authenticated-orcid":false,"given":"Zeynep","family":"Akata","sequence":"additional","affiliation":[{"name":"Explainable Machine Learning Group at the University of T&#x00FC;bingen, T&#x00FC;bingen, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1732-9198","authenticated-orcid":false,"given":"Samuel","family":"Albanie","sequence":"additional","affiliation":[{"name":"Department of Engineering at the University of Cambridge, Cambridge, U.K."}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref13","first-page":"3","article-title":"Audio events detection based highlights extraction from baseball, golf and soccer games in a unified framework","author":"xiong","year":"0","journal-title":"Proc Int Conf Multimedia Expo"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-11018-5_62"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-2227"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2445"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2006.1661400"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-68780-9_26"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1121\/1.2750160"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1145\/3387164"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462034"},{"key":"ref52","first-page":"59","article-title":"Acoustic event search with an onomatopoeic query: Measuring distance between onomatopoeic words and sounds","author":"ikawa","year":"0","journal-title":"Proc Detection Classification Acoust Scenes Events Workshop"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1145\/2502081.2502245"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01231-1_40"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"ref54","article-title":"See, hear, and read: Deep aligned representations","author":"aytar","year":"2017"},{"key":"ref17","article-title":"DCASE 2017 challenge setup: Tasks, datasets and baseline system","author":"mesaros","year":"0","journal-title":"Proc Workshop Detection Classification Acoust Scenes Events"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2015.2428998"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/WASPAA.2015.7336899"},{"key":"ref18","article-title":"TUT acoustic scenes 2017, Dev. Dataset","author":"mesaros","year":"2017"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1145\/1460096.1460115"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/ICME.2002.1035789"},{"key":"ref90","first-page":"65","article-title":"METEOR: An automatic metric for MT evaluation with improved correlation with human judgments","author":"banerjee","year":"0","journal-title":"Proc ACL Workshop"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00175"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2018.2832602"},{"key":"ref89","article-title":"Speech-to-text API","year":"0"},{"key":"ref48","article-title":"Nels-never-ending learner of sounds","author":"elizalde","year":"0"},{"key":"ref47","article-title":"CLIP4Clip: An empirical study of clip for end to end video clip retrieval","author":"luo","year":"2021"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1155\/2013\/972438"},{"key":"ref86","article-title":"On the variance of the adaptive learning rate and beyond","author":"liu","year":"2019"},{"key":"ref41","article-title":"Audio-based near-duplicate video retrieval with audio similarity learning","author":"avgoustinakis","year":"2020"},{"key":"ref85","first-page":"9597","article-title":"Lookahead optimizer: K steps forward, 1 step back","author":"zhang","year":"0","journal-title":"Proc Neural Inf Process Syst"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01138"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-20890-5_3"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00054"},{"key":"ref87","article-title":"Ranger optimiser","year":"0"},{"key":"ref49","first-page":"iv-4108","article-title":"Semantic-audio retrieval","author":"slaney","year":"0","journal-title":"Proc IEEE Int Conf Acoust Speech Signal Process"},{"key":"ref8","first-page":"119","article-title":"AudioCaps: Generating captions for audios in the wild","author":"kim","year":"0","journal-title":"Proc Conf North Amer Chapter Assoc Comput Linguistics Hum Lang Technol"},{"key":"ref7","first-page":"214","article-title":"Multi-modal transformer for video retrieval","author":"gabeur","year":"0","journal-title":"Proc Eur Conf Comput Vis"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9052990"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461524"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1145\/3206025.3206064"},{"key":"ref6","article-title":"Use what you have: Video retrieval using representations from collaborative experts","author":"liu","year":"0","journal-title":"Proc Brit Mach Vis Conf"},{"key":"ref5","first-page":"1","article-title":"Content-based retrieval of environmental sounds by multiresolution analysis","author":"lallemand","year":"0","journal-title":"Proc SMC"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2017.2723009"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.243"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2012-556"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01232"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00675"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref35","article-title":"Audio captioning transformer","author":"mei","year":"2021"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01216-8_12"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ISM.2020.00014"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v29i1.9512"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1117\/12.290336"},{"key":"ref36","article-title":"CL4AC: A contrastive loss for audio captioning","author":"liu","year":"2021"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2017.2729019"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952132"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.515"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.571"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413982"},{"key":"ref77","article-title":"Deep residual learning for image recognition","author":"he","year":"2015"},{"key":"ref32","article-title":"Audio captioning using pre-trained large-scale language model guided by audio-based similar caption retrieval","author":"koizumi","year":"2020"},{"key":"ref76","article-title":"YouTube-8M: A large-scale video classification benchmark","author":"abu-el-haija","year":"2016"},{"key":"ref2","article-title":"Learning a text-video embedding from incomplete and heterogeneous data","author":"miech","year":"2018"},{"key":"ref1","article-title":"Word2VisualVec: Image and video to sentence matching by visual feature prediction","author":"dong","year":"2016"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2007.366657"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/93.556537"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.83"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053174"},{"key":"ref73","article-title":"Youdescribe","year":"2013"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414640"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2930913"},{"key":"ref68","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"devlin","year":"2018"},{"key":"ref23","article-title":"Multi-level attention model for weakly supervised audio classification","author":"yu","year":"2018"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.572"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3030497"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2731"},{"key":"ref69","article-title":"Project page","year":"0"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1145\/2733373.2806390"},{"key":"ref64","article-title":"Efficient estimation of word representations in vector space","author":"mikolov","year":"2013"},{"key":"ref63","article-title":"Query-graph with cross-gating attention model for text-to-audio grounding","author":"tang","year":"2021"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461392"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00177"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.33682\/w13e-5v06"},{"key":"ref65","first-page":"2579","article-title":"Visualizing data using t-SNE","volume":"9","author":"maaten","year":"2008","journal-title":"J Mach Learn Res"},{"key":"ref28","article-title":"DCASE2020 challenge task 6: Automated audio captioning","year":"2020"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/WASPAA.2017.8170058"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682377"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01261-8_5"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414834"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682632"}],"container-title":["IEEE Transactions on Multimedia"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6046\/10016790\/09707629.pdf?arnumber=9707629","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,7]],"date-time":"2023-08-07T18:19:56Z","timestamp":1691432396000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9707629\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"references-count":90,"URL":"https:\/\/doi.org\/10.1109\/tmm.2022.3149712","relation":{},"ISSN":["1520-9210","1941-0077"],"issn-type":[{"value":"1520-9210","type":"print"},{"value":"1941-0077","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023]]}}}