{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,7,20]],"date-time":"2025-07-20T04:08:55Z","timestamp":1752984535979,"version":"3.37.3"},"reference-count":21,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,6,23]],"date-time":"2021-06-23T00:00:00Z","timestamp":1624406400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2021,6,23]],"date-time":"2021-06-23T00:00:00Z","timestamp":1624406400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,6,23]],"date-time":"2021-06-23T00:00:00Z","timestamp":1624406400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100002341","name":"Academy of Finland","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100002341","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,6,23]]},"DOI":"10.1109\/euvip50544.2021.9484036","type":"proceedings-article","created":{"date-parts":[[2021,7,20]],"date-time":"2021-07-20T20:33:18Z","timestamp":1626813198000},"page":"1-6","source":"Crossref","is-referenced-by-count":5,"title":["Leveraging Category Information for Single-Frame Visual Sound Source Separation"],"prefix":"10.1109","author":[{"given":"Lingyu","family":"Zhu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Esa","family":"Rahtu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00097"},{"key":"ref11","first-page":"35","article-title":"Learning to separate object sounds by watching unlabeled video","author":"gao","year":"2018","journal-title":"Proceedings of the European Conference on Computer Vision (ECCV)"},{"key":"ref12","first-page":"631","article-title":"Audio-visual scene analysis with self-supervised multisensory features","author":"owens","year":"2018","journal-title":"Proceedings of the European Conference on Computer Vision (ECCV)"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00398"},{"key":"ref14","first-page":"813","article-title":"Audio vision: Using audio-visual synchrony to locate sounds","author":"hershey","year":"2000","journal-title":"Advances in neural information processing systems"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2005.274"},{"key":"ref16","first-page":"435","article-title":"Objects that sound","author":"arandjelovic","year":"2018","journal-title":"Proceedings of the European Conference on Computer Vision (ECCV)"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-018-1083-5"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref19","first-page":"234","article-title":"U-net: Convolutional networks for biomedical image segmentation","author":"ronneberger","year":"2015","journal-title":"International Conference on Medical Image Computing and Computer-Assisted Intervention"},{"key":"ref4","doi-asserted-by":"crossref","DOI":"10.1145\/3197517.3201357","article-title":"Looking to listen at the cocktail party: A speaker-independent audio-visual model for speech separation","author":"ephrat","year":"2018"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA40945.2020.9197008"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.73"},{"key":"ref5","first-page":"570","article-title":"The sound of pixels","author":"zhao","year":"2018","journal-title":"Proceedings of the European Conference on Computer Vision (ECCV)"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01049"},{"key":"ref7","article-title":"Visually guided sound source separation using cascaded opponent filter network","author":"zhu","year":"2020","journal-title":"Proceedings of the Asian Conference on Computer Vision"},{"key":"ref2","article-title":"Learning representations from audio-visual spatial alignment","volume":"33","author":"morgado","year":"2020","journal-title":"Advances in neural information processing systems"},{"key":"ref1","article-title":"Self-supervised learning of audio-visual objects from video","author":"afouras","year":"2020","journal-title":"European Conference on Computer Vision"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00182"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00474"},{"key":"ref21","first-page":"2579","article-title":"Visualizing data using t-sne","volume":"9","author":"van der maaten","year":"2008","journal-title":"Journal of Machine Learning Research"}],"event":{"name":"2021 9th European Workshop on Visual Information Processing (EUVIP)","start":{"date-parts":[[2021,6,23]]},"location":"Paris, France","end":{"date-parts":[[2021,6,25]]}},"container-title":["2021 9th European Workshop on Visual Information Processing (EUVIP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9483885\/9483956\/09484036.pdf?arnumber=9484036","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,5,10]],"date-time":"2022-05-10T15:48:08Z","timestamp":1652197688000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9484036\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,6,23]]},"references-count":21,"URL":"https:\/\/doi.org\/10.1109\/euvip50544.2021.9484036","relation":{},"subject":[],"published":{"date-parts":[[2021,6,23]]}}}