{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T14:51:50Z","timestamp":1784904710598,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":32,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100011178","name":"Institute of Automation, Chinese Academy of Sciences","doi-asserted-by":"publisher","award":["Grant No. E411230101"],"award-info":[{"award-number":["Grant No. E411230101"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100011178","id-type":"DOI","asserted-by":"publisher"}]},{"name":"the National Natural Science Foundation of China","award":["Grant No. 62372453"],"award-info":[{"award-number":["Grant No. 62372453"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681503","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:27Z","timestamp":1729925967000},"page":"985-993","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":9,"title":["CACE-Net: Co-guidance Attention and Contrastive Enhancement for Effective Audio-Visual Event Localization"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9100-328X","authenticated-orcid":false,"given":"Xiang","family":"He","sequence":"first","affiliation":[{"name":"Brain-inspired Cognitive Intelligence Lab, Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-3793-8603","authenticated-orcid":false,"given":"Xiangxi","family":"Liu","sequence":"additional","affiliation":[{"name":"Brain-inspired Cognitive Intelligence Lab, Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0161-9801","authenticated-orcid":false,"given":"Yang","family":"Li","sequence":"additional","affiliation":[{"name":"Brain-inspired Cognitive Intelligence Lab, Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0593-8650","authenticated-orcid":false,"given":"Dongcheng","family":"Zhao","sequence":"additional","affiliation":[{"name":"Brain-inspired Cognitive Intelligence Lab, Institute of Automation, Chinese Academy of Sciences &amp; Center for Long-term Artificial Intelligence, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4069-2107","authenticated-orcid":false,"given":"Guobin","family":"Shen","sequence":"additional","affiliation":[{"name":"Brain-inspired Cognitive Intelligence Lab, Institute of Automation, Chinese Academy of Sciences &amp; Center for Long-term Artificial Intelligence, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8565-549X","authenticated-orcid":false,"given":"Qingqun","family":"Kong","sequence":"additional","affiliation":[{"name":"Brain-inspired Cognitive Intelligence Lab, Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5702-9451","authenticated-orcid":false,"given":"Xin","family":"Yang","sequence":"additional","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9595-9091","authenticated-orcid":false,"given":"Yi","family":"Zeng","sequence":"additional","affiliation":[{"name":"Brain-inspired Cognitive Intelligence Lab, Institute of Automation, Chinese Academy of Sciences &amp; Center for Long-term Artificial Intelligence, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Soundnet: Learning sound representations from unlabeled video. In Advances in Neural Information Processing Systems (NeurIPS).","author":"Aytar Yusuf","year":"2016","unstructured":"Yusuf Aytar, Carl Vondrick, and Antonio Torralba. 2016. Soundnet: Learning sound representations from unlabeled video. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_1_2_1","volume-title":"hear, and read: Deep aligned representations. arXiv preprint arXiv:1706.00932","author":"Aytar Yusuf","year":"2017","unstructured":"Yusuf Aytar, Carl Vondrick, and Antonio Torralba. 2017. See, hear, and read: Deep aligned representations. arXiv preprint arXiv:1706.00932 (2017)."},{"key":"e_1_3_2_1_3_1","volume-title":"Merging the senses into a robust percept. Trends in cognitive sciences","author":"Ernst Marc O","year":"2004","unstructured":"Marc O Ernst and Heinrich H B\u00fclthoff. 2004. Merging the senses into a robust percept. Trends in cognitive sciences, Vol. 8, 4 (2004), 162--169."},{"key":"e_1_3_2_1_4_1","volume-title":"Css-net: A consistent segment selection network for audio-visual event localization","author":"Feng Fan","year":"2023","unstructured":"Fan Feng, Yue Ming, Nannan Hu, Hui Yu, and Yuanan Liu. 2023. Css-net: A consistent segment selection network for audio-visual event localization. IEEE Transactions on Multimedia (2023)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01049"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01047"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747631"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_11_1","volume-title":"Jort F Gemmeke, Aren Jansen, R Channing Moore, Manoj Plakal, Devin Platt, Rif A Saurous, Bryan Seybold, et al.","author":"Hershey Shawn","year":"2017","unstructured":"Shawn Hershey, Sourish Chaudhuri, Daniel PW Ellis, Jort F Gemmeke, Aren Jansen, R Channing Moore, Manoj Plakal, Devin Platt, Rif A Saurous, Bryan Seybold, et al. 2017. CNN architectures for large-scale audio classification. In 2017 ieee international conference on acoustics, speech and signal processing (icassp). IEEE, 131--135."},{"key":"e_1_3_2_1_12_1","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition Workshops. 2528--2531","author":"Hori Chiori","year":"2018","unstructured":"Chiori Hori, Takaaki Hori, Gordon Wichern, Jue Wang, Teng-Yok Lee, Anoop Cherian, and Tim K Marks. 2018. Multimodal attention for fusion of audio and spatiotemporal features for video description. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition Workshops. 2528--2531."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00947"},{"key":"e_1_3_2_1_14_1","first-page":"10077","article-title":"Discriminative sounding objects localization via self-supervised audiovisual matching","volume":"33","author":"Hu Di","year":"2020","unstructured":"Di Hu, Rui Qian, Minyue Jiang, Xiao Tan, Shilei Wen, Errui Ding, Weiyao Lin, and Dejing Dou. 2020. Discriminative sounding objects localization via self-supervised audiovisual matching. Advances in Neural Information Processing Systems, Vol. 33 (2020), 10077--10087.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00745"},{"key":"e_1_3_2_1_16_1","volume-title":"Supervised contrastive learning. Advances in neural information processing systems","author":"Khosla Prannay","year":"2020","unstructured":"Prannay Khosla, Piotr Teterwak, Chen Wang, Aaron Sarna, Yonglong Tian, Phillip Isola, Aaron Maschinot, Ce Liu, and Dilip Krishnan. 2020. Supervised contrastive learning. Advances in neural information processing systems, Vol. 33 (2020), 18661--18673."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3030497"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683226"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.12319"},{"key":"e_1_3_2_1_20_1","volume-title":"Learning audio-visual representations with active contrastive coding. arXiv preprint arXiv:2009.09805","author":"Ma Shuang","year":"2020","unstructured":"Shuang Ma, Zhaoyang Zeng, Daniel McDuff, and Yale Song. 2020. Learning audio-visual representations with active contrastive coding. arXiv preprint arXiv:2009.09805, Vol. 2 (2020)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV56688.2023.00513"},{"key":"e_1_3_2_1_22_1","volume-title":"Perceptual inference, learning, and attention in a multisensory world. Annual review of neuroscience","author":"Noppeney Uta","year":"2021","unstructured":"Uta Noppeney. 2021. Perceptual inference, learning, and attention in a multisensory world. Annual review of neuroscience, Vol. 44 (2021), 449--473."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46448-0_48"},{"key":"e_1_3_2_1_24_1","volume-title":"International conference on machine learning. PMLR, 8748--8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748--8763."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3161174"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"crossref","unstructured":"Olga Russakovsky Jia Deng Hao Su Jonathan Krause Sanjeev Satheesh Sean Ma Zhiheng Huang Andrej Karpathy Aditya Khosla Michael Bernstein et al. 2015. Imagenet large scale visual recognition challenge. International journal of computer vision Vol. 115 (2015) 211--252.","DOI":"10.1007\/s11263-015-0816-y"},{"key":"e_1_3_2_1_27_1","volume-title":"Very Deep Convolutional Networks for Large-Scale Image Recognition. In 3rd International Conference on Learning Representations, ICLR","author":"Simonyan Karen","year":"2015","unstructured":"Karen Simonyan and Andrew Zisserman. 2015. Very Deep Convolutional Networks for Large-Scale Image Recognition. In 3rd International Conference on Learning Representations, ICLR 2015."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01216-8_16"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00639"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01936"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413581"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00833"}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681503","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681503","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:57:48Z","timestamp":1750294668000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681503"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":32,"alternative-id":["10.1145\/3664647.3681503","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681503","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}