{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T18:40:50Z","timestamp":1776883250687,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":53,"publisher":"ACM","funder":[{"name":"Sony AI"},{"name":"Commonwealth Bank of Australia"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3754919","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:47:18Z","timestamp":1761374838000},"page":"7510-7518","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["CCStereo: Audio-Visual Contextual and Contrastive Learning for Binaural Audio Generation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8983-2895","authenticated-orcid":false,"given":"Yuanhong","family":"Chen","sequence":"first","affiliation":[{"name":"Australian Institute for Machine Learning, University of Adelaide, Adelaide, SA, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5389-2346","authenticated-orcid":false,"given":"Kazuki","family":"Shimada","sequence":"additional","affiliation":[{"name":"Sony AI, Tokyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-3642-820X","authenticated-orcid":false,"given":"Christian","family":"Simon","sequence":"additional","affiliation":[{"name":"Sony Group Corporation, Tokyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-0296-4390","authenticated-orcid":false,"given":"Yukara","family":"Ikemiya","sequence":"additional","affiliation":[{"name":"Sony AI, Tokyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4277-0164","authenticated-orcid":false,"given":"Takashi","family":"Shibuya","sequence":"additional","affiliation":[{"name":"Sony AI, Tokyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6806-6140","authenticated-orcid":false,"given":"Yuki","family":"Mitsufuji","sequence":"additional","affiliation":[{"name":"Sony AI, Sony Group Corporation, New York, NY, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447648"},{"key":"e_1_3_2_2_2_1","volume-title":"Multimodal machine learning: A survey and taxonomy","author":"Baltru\u0161aitis Tadas","year":"2018","unstructured":"Tadas Baltru\u0161aitis, Chaitanya Ahuja, and Louis-Philippe Morency. 2018. Multimodal machine learning: A survey and taxonomy. IEEE transactions on pattern analysis and machine intelligence 41, 2 (2018), 423--443."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"crossref","unstructured":"Honglie Chen Weidi Xie Triantafyllos Afouras Arsha Nagrani Andrea Vedaldi and Andrew Zisserman. 2021. Localizing visual sounds the hard way. In CVPR. 16867--16876.","DOI":"10.1109\/CVPR46437.2021.01659"},{"key":"e_1_3_2_2_4_1","volume-title":"International conference on machine learning. PMLR, 1597--1607","author":"Chen Ting","year":"2020","unstructured":"Ting Chen, Simon Kornblith, Mohammad Norouzi, and Geoffrey Hinton. 2020. A simple framework for contrastive learning of visual representations. In International conference on machine learning. PMLR, 1597--1607."},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01549"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"crossref","unstructured":"Yuanhong Chen Yuyuan Liu Hu Wang Fengbei Liu Chong Wang Helen Frazer and Gustavo Carneiro. 2024. Unraveling Instance Associations: A Closer Look for Audio-Visual Segmentation. In CVPR. 26497--26507.","DOI":"10.1109\/CVPR52733.2024.02502"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1088\/2516-1091\/acc2fe"},{"key":"e_1_3_2_2_8_1","volume-title":"Imagenet: A large-scale hierarchical image database. In CVPR. Ieee, 248--255.","author":"Deng Jia","year":"2009","unstructured":"Jia Deng,Wei Dong, Richard Socher, Li-Jia Li, Kai Li, and Li Fei-Fei. 2009. Imagenet: A large-scale hierarchical image database. In CVPR. Ieee, 248--255."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.608"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11330"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"crossref","unstructured":"Ruohan Gao and Kristen Grauman. 2019. 2.5 d visual sound. In CVPR. 324--333.","DOI":"10.1109\/CVPR.2019.00041"},{"key":"e_1_3_2_2_12_1","volume-title":"Geometry-aware multitask learning for binaural audio generation from video. BMVC","author":"Garg Rishabh","year":"2021","unstructured":"Rishabh Garg, Ruohan Gao, and Kristen Grauman. 2021. Geometry-aware multitask learning for binaural audio generation from video. BMVC (2021)."},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.265"},{"key":"e_1_3_2_2_14_1","volume-title":"Pooling methods in deep neural networks, a review. arXiv preprint arXiv:2009.07485","author":"Gholamalinezhad Hossein","year":"2020","unstructured":"Hossein Gholamalinezhad and Hossein Khosravi. 2020. Pooling methods in deep neural networks, a review. arXiv preprint arXiv:2009.07485 (2020)."},{"key":"e_1_3_2_2_15_1","volume-title":"Audio Engineering Society Conference: 8th International Conference: The Sound of Audio. Audio Engineering Society.","author":"Griesinger David","year":"1990","unstructured":"David Griesinger. 1990. Binaural techniques for music reproduction. In Audio Engineering Society Conference: 8th International Conference: The Sound of Audio. Audio Engineering Society."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1984.1164317"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"crossref","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep residual learning for image recognition. In CVPR. 770--778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_2_19_1","volume-title":"Jort F Gemmeke, Aren Jansen, R Channing Moore, Manoj Plakal, Devin Platt, Rif A Saurous, Bryan Seybold, et al.","author":"Hershey Shawn","year":"2017","unstructured":"Shawn Hershey, Sourish Chaudhuri, Daniel PW Ellis, Jort F Gemmeke, Aren Jansen, R Channing Moore, Manoj Plakal, Devin Platt, Rif A Saurous, Bryan Seybold, et al. 2017. CNN architectures for large-scale audio classification. In ICASSP. IEEE, 131--135."},{"key":"e_1_3_2_2_20_1","first-page":"6840","article-title":"Denoising diffusion probabilistic models","volume":"33","author":"Ho Jonathan","year":"2020","unstructured":"Jonathan Ho, Ajay Jain, and Pieter Abbeel. 2020. Denoising diffusion probabilistic models. NeurIPS 33 (2020), 6840--6851.","journal-title":"NeurIPS"},{"key":"e_1_3_2_2_21_1","volume-title":"Binaural sound reduces reaction time in a virtual reality search task","author":"Hoeg Emil R","unstructured":"Emil R Hoeg, Lynda J Gerry, Lui Thomsen, Niels C Nilsson, and Stefania Serafin. 2017. Binaural sound reduces reaction time in a virtual reality search task. In SIVE. IEEE, 1--4."},{"key":"e_1_3_2_2_22_1","unstructured":"Xixi Hu Ziyang Chen and Andrew Owens. 2022. Mix and localize: Localizing sound sources in mixtures. In CVPR. 10483--10492."},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.167"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00453"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-92185-9_46"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3478513.3480560"},{"key":"e_1_3_2_2_27_1","volume-title":"A survey of multi-view representation learning","author":"Li Yingming","year":"2018","unstructured":"Yingming Li, Ming Yang, and Zhongfei Zhang. 2018. A survey of multi-view representation learning. IEEE transactions on knowledge and data engineering 31, 10 (2018), 1863--1883."},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2024.111814"},{"key":"e_1_3_2_2_29_1","unstructured":"Zhaojian Li Bin Zhao and Yuan Yuan. 2024. Cyclic Learning for Binaural Audio Generation and Localization. In CVPR. 26669--26678."},{"key":"e_1_3_2_2_30_1","volume-title":"Visually Guided Binaural Audio Generation with Cross-Modal Consistency. In ICASSP 2024--2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 7980--7984","author":"Liu Miao","year":"2024","unstructured":"Miao Liu, Jing Wang, Xinyuan Qian, and Xiang Xie. 2024. Visually Guided Binaural Audio Generation with Cross-Modal Consistency. In ICASSP 2024--2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 7980--7984."},{"key":"e_1_3_2_2_31_1","volume-title":"A Closer Look at Weakly-Supervised Audio-Visual Source Localization. arXiv preprint arXiv:2209.09634","author":"Mo Shentong","year":"2022","unstructured":"Shentong Mo and Pedro Morgado. 2022. A Closer Look at Weakly-Supervised Audio-Visual Source Localization. arXiv preprint arXiv:2209.09634 (2022)."},{"key":"e_1_3_2_2_32_1","volume-title":"Localizing visual sounds the easy way","author":"Mo Shentong","unstructured":"Shentong Mo and Pedro Morgado. 2022. Localizing visual sounds the easy way. In ECCV. Springer, 218--234."},{"key":"e_1_3_2_2_33_1","volume-title":"Self-supervised generation of spatial audio for 360 video. NeurIPS 31","author":"Morgado Pedro","year":"2018","unstructured":"Pedro Morgado, Nuno Nvasconcelos, Timothy Langlois, and Oliver Wang. 2018. Self-supervised generation of spatial audio for 360 video. NeurIPS 31 (2018)."},{"key":"e_1_3_2_2_34_1","volume-title":"Batch-instance normalization for adaptively style-invariant neural networks. NeurIPS 31","author":"Nam Hyeonseob","year":"2018","unstructured":"Hyeonseob Nam and Hyo-Eun Kim. 2018. Batch-instance normalization for adaptively style-invariant neural networks. NeurIPS 31 (2018)."},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV51458.2022.00221"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"crossref","unstructured":"Taesung Park Ming-Yu Liu Ting-Chun Wang and Jun-Yan Zhu. 2019. Semantic image synthesis with spatially-adaptive normalization. In CVPR. 2337--2346.","DOI":"10.1109\/CVPR.2019.00244"},{"key":"e_1_3_2_2_37_1","volume-title":"Fernando Torre, and Yaser Sheikh.","author":"Richard Alexander","year":"2021","unstructured":"Alexander Richard, Dejan Markovic, Israel D Gebru, Steven Krenn, Gladstone Alexander Butler, Fernando Torre, and Yaser Sheikh. 2021. Neural synthesis of binaural speech from mono audio. In ICLR."},{"key":"e_1_3_2_2_38_1","volume-title":"U-net: Convolutional networks for biomedical image segmentation","author":"Ronneberger Olaf","year":"2015","unstructured":"Olaf Ronneberger, Philipp Fischer, and Thomas Brox. 2015. U-net: Convolutional networks for biomedical image segmentation. In MICCAI. Springer, 234--241."},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00125"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00787"},{"key":"e_1_3_2_2_41_1","volume-title":"Attention is all you need. NeurIPS","author":"Vaswani A","year":"2017","unstructured":"A Vaswani. 2017. Attention is all you need. NeurIPS (2017)."},{"key":"e_1_3_2_2_42_1","volume-title":"Semantic image synthesis via diffusion models. arXiv preprint arXiv:2207.00050","author":"Wang Weilun","year":"2022","unstructured":"Weilun Wang, Jianmin Bao, Wengang Zhou, Dongdong Chen, Dong Chen, Lu Yuan, and Houqiang Li. 2022. Semantic image synthesis via diffusion models. arXiv preprint arXiv:2207.00050 (2022)."},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01271"},{"key":"e_1_3_2_2_44_1","volume-title":"A survey on multi-view learning. arXiv preprint arXiv:1304.5634","author":"Xu Chang","year":"2013","unstructured":"Chang Xu, Dacheng Tao, and Chao Xu. 2013. A survey on multi-view learning. arXiv preprint arXiv:1304.5634 (2013)."},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"crossref","unstructured":"Xudong Xu Hang Zhou Ziwei Liu Bo Dai XiaogangWang and Dahua Lin. 2021. Visually informed binaural audio generation without binaural audios. In CVPR. 15485--15494.","DOI":"10.1109\/CVPR46437.2021.01523"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i01.5361"},{"key":"e_1_3_2_2_47_1","volume-title":"Ip-adapter: Text compatible image prompt adapter for text-to-image diffusion models. arXiv preprint arXiv:2308.06721","author":"Ye Hu","year":"2023","unstructured":"Hu Ye, Jun Zhang, Sibo Liu, Xiao Han, and Wei Yang. 2023. Ip-adapter: Text compatible image prompt adapter for text-to-image diffusion models. arXiv preprint arXiv:2308.06721 (2023)."},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"crossref","unstructured":"Wen Zhang and Jie Shao. 2021. Multi-attention audio-visual fusion network for audio spatialization. In ICMR. 394--401.","DOI":"10.1145\/3460426.3463624"},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"crossref","unstructured":"Hang Zhao Chuang Gan Andrew Rouditchenko Carl Vondrick Josh McDermott and Antonio Torralba. 2018. The sound of pixels. In ECCV. 570--586.","DOI":"10.1007\/978-3-030-01246-5_35"},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41598-024-51258-6"},{"key":"e_1_3_2_2_52_1","volume-title":"Sepstereo: Visually guided stereophonic audio generation by associating source separation","author":"Zhou Hang","year":"2020","unstructured":"Hang Zhou, Xudong Xu, Dahua Lin, Xiaogang Wang, and Ziwei Liu. 2020. Sepstereo: Visually guided stereophonic audio generation by associating source separation. In ECCV. Springer, 52--69."},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00261"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3754919","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:18:19Z","timestamp":1765340299000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3754919"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":53,"alternative-id":["10.1145\/3746027.3754919","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3754919","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}