{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:41:23Z","timestamp":1755823283248,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":61,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62272178, and 62276109"],"award-info":[{"award-number":["62272178, and 62276109"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3613805","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:12Z","timestamp":1698391632000},"page":"261-270","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["SCLAV: Supervised Cross-modal Contrastive Learning for Audio-Visual Coding"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8735-4893","authenticated-orcid":false,"given":"Chao","family":"Sun","sequence":"first","affiliation":[{"name":"Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0960-4447","authenticated-orcid":false,"given":"Min","family":"Chen","sequence":"additional","affiliation":[{"name":"South China University of Technology &amp; Pazhou Laboratory, Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-5144-031X","authenticated-orcid":false,"given":"Jialiang","family":"Cheng","sequence":"additional","affiliation":[{"name":"Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8493-6495","authenticated-orcid":false,"given":"Han","family":"Liang","sequence":"additional","affiliation":[{"name":"Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6441-1162","authenticated-orcid":false,"given":"Chuanbo","family":"Zhu","sequence":"additional","affiliation":[{"name":"Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7368-1677","authenticated-orcid":false,"given":"Jincai","family":"Chen","sequence":"additional","affiliation":[{"name":"Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01032"},{"key":"e_1_3_2_1_2_1","first-page":"24206","article-title":"Vatt: Transformers for multimodal self-supervised learning from raw video, audio and text","volume":"34","author":"Akbari Hassan","year":"2021","unstructured":"Hassan Akbari, Liangzhe Yuan, Rui Qian, Wei-Hong Chuang, Shih-Fu Chang, Yin Cui, and Boqing Gong. 2021. Vatt: Transformers for multimodal self-supervised learning from raw video, audio and text. Advances in Neural Information Processing Systems 34 (2021), 24206--24221.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_3_1","first-page":"25","article-title":"Self-supervised multimodal versatile networks","volume":"33","author":"Alayrac Jean-Baptiste","year":"2020","unstructured":"Jean-Baptiste Alayrac, Adria Recasens, Rosalia Schneider, Relja Arandjelovi\u0107, Jason Ramapuram, Jeffrey De Fauw, Lucas Smaira, Sander Dieleman, and Andrew Zisserman. 2020. Self-supervised multimodal versatile networks. Advances in Neural Information Processing Systems 33 (2020), 25--37.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_4_1","first-page":"9758","article-title":"Self-supervised learning by cross-modal audiovideo clustering","volume":"33","author":"Alwassel Humam","year":"2020","unstructured":"Humam Alwassel, Dhruv Mahajan, Bruno Korbar, Lorenzo Torresani, Bernard Ghanem, and Du Tran. 2020. Self-supervised learning by cross-modal audiovideo clustering. Advances in Neural Information Processing Systems 33 (2020), 9758--9770.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.73"},{"key":"e_1_3_2_1_6_1","volume-title":"International Conference on Machine Learning. PMLR, 1298--1312","author":"Baevski Alexei","year":"2022","unstructured":"Alexei Baevski, Wei-Ning Hsu, Qiantong Xu, Arun Babu, Jiatao Gu, and Michael Auli. 2022. Data2vec: A general framework for self-supervised learning in speech, vision and language. In International Conference on Machine Learning. PMLR, 1298--1312."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2022.03.001"},{"key":"e_1_3_2_1_8_1","volume-title":"Unsupervised learning of visual features by contrasting cluster assignments. Advances in neural information processing systems 33","author":"Caron Mathilde","year":"2020","unstructured":"Mathilde Caron, Ishan Misra, Julien Mairal, Priya Goyal, Piotr Bojanowski, and Armand Joulin. 2020. Unsupervised learning of visual features by contrasting cluster assignments. Advances in neural information processing systems 33 (2020), 9912--9924."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00791"},{"key":"e_1_3_2_1_11_1","volume-title":"A simple framework for contrastive learning of visual representations. ICML. arXiv preprint arXiv:2002.05709","author":"Chen Ting","year":"2020","unstructured":"Ting Chen, Simon Kornblith, Mohammad Norouzi, and Geoffrey Hinton. 2020. A simple framework for contrastive learning of visual representations. ICML. arXiv preprint arXiv:2002.05709 (2020)."},{"key":"e_1_3_2_1_12_1","volume-title":"International conference on machine learning. PMLR","author":"Frosst Nicholas","year":"2019","unstructured":"Nicholas Frosst, Nicolas Papernot, and Geoffrey Hinton. 2019. Analyzing and improving representations with the soft nearest neighbor loss. In International conference on machine learning. PMLR, 2012-2020."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548411"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747631"},{"key":"e_1_3_2_1_15_1","first-page":"5679","article-title":"Self-supervised co-training for video representation learning","volume":"33","author":"Han Tengda","year":"2020","unstructured":"Tengda Han,Weidi Xie, and Andrew Zisserman. 2020. Self-supervised co-training for video representation learning. Advances in Neural Information Processing Systems 33 (2020), 5679--5690.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"e_1_3_2_1_17_1","volume-title":"International conference on machine learning. PMLR, 4182--4192","author":"Henaff Olivier","year":"2020","unstructured":"Olivier Henaff. 2020. Data-efficient image recognition with contrastive predictive coding. In International conference on machine learning. PMLR, 4182--4192."},{"key":"e_1_3_2_1_18_1","volume-title":"Jort F Gemmeke, Aren Jansen, R Channing Moore, Manoj Plakal, Devin Platt, Rif A Saurous, Bryan Seybold, et al.","author":"Hershey Shawn","year":"2017","unstructured":"Shawn Hershey, Sourish Chaudhuri, Daniel PW Ellis, Jort F Gemmeke, Aren Jansen, R Channing Moore, Manoj Plakal, Devin Platt, Rif A Saurous, Bryan Seybold, et al. 2017. CNN architectures for large-scale audio classification. In 2017 ieee international conference on acoustics, speech and signal processing (icassp). IEEE, 131--135."},{"key":"e_1_3_2_1_19_1","volume-title":"Learning deep representations by mutual information estimation and maximization. arXiv preprint arXiv:1808.06670","author":"Hjelm R Devon","year":"2018","unstructured":"R Devon Hjelm, Alex Fedorov, Samuel Lavoie-Marchildon, Karan Grewal, Phil Bachman, Adam Trischler, and Yoshua Bengio. 2018. Learning deep representations by mutual information estimation and maximization. arXiv preprint arXiv:1808.06670 (2018)."},{"key":"e_1_3_2_1_20_1","first-page":"17081","article-title":"Contrastive learning with adversarial examples","volume":"33","author":"Ho Chih-Hui","year":"2020","unstructured":"Chih-Hui Ho and Nuno Nvasconcelos. 2020. Contrastive learning with adversarial examples. Advances in Neural Information Processing Systems 33 (2020), 17081--17093.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_21_1","volume-title":"International conference on machine learning. PMLR, 2459--2468","author":"Kamnitsas Konstantinos","year":"2018","unstructured":"Konstantinos Kamnitsas, Daniel Castro, Loic Le Folgoc, Ian Walker, Ryutaro Tanno, Daniel Rueckert, Ben Glocker, Antonio Criminisi, and Aditya Nori. 2018. Semi-supervised learning via compact latent space clustering. In International conference on machine learning. PMLR, 2459--2468."},{"key":"e_1_3_2_1_22_1","volume-title":"Supervised contrastive learning. Advances in neural information processing systems 33","author":"Khosla Prannay","year":"2020","unstructured":"Prannay Khosla, Piotr Teterwak, ChenWang, Aaron Sarna, Yonglong Tian, Phillip Isola, Aaron Maschinot, Ce Liu, and Dilip Krishnan. 2020. Supervised contrastive learning. Advances in neural information processing systems 33 (2020), 18661--18673."},{"key":"e_1_3_2_1_23_1","volume-title":"The local rademacher complexity of lp-norm multiple kernel learning. Advances in Neural Information Processing Systems 24","author":"Kloft Marius","year":"2011","unstructured":"Marius Kloft and Gilles Blanchard. 2011. The local rademacher complexity of lp-norm multiple kernel learning. Advances in Neural Information Processing Systems 24 (2011)."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i2.20028"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683226"},{"key":"e_1_3_2_1_26_1","volume-title":"Proceedings of the Asian Conference on Computer Vision.","author":"Lin Yan-Bo","year":"2020","unstructured":"Yan-Bo Lin and Yu-Chiang Frank Wang. 2020. Audiovisual transformer with instance attention for audio-visual event localization. In Proceedings of the Asian Conference on Computer Vision."},{"key":"e_1_3_2_1_27_1","volume-title":"Proceedings of the Asian Conference on Computer Vision (ACCV).","author":"Lin Yan-Bo","year":"2020","unstructured":"Yan-Bo Lin and Yu-Chiang Frank Wang. 2020. Audiovisual Transformer with Instance Attention for Audio-Visual Event Localization. In Proceedings of the Asian Conference on Computer Vision (ACCV)."},{"volume-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 6707--6717","author":"Misra Ishan","key":"e_1_3_2_1_28_1","unstructured":"Ishan Misra and Laurens van der Maaten. 2020. Self-supervised learning of pretext-invariant representations. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 6707--6717."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01274"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01229"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548397"},{"key":"e_1_3_2_1_32_1","volume-title":"Representation learning with contrastive predictive coding. arXiv preprint arXiv:1807.03748","author":"van den Oord Aaron","year":"2018","unstructured":"Aaron van den Oord, Yazhe Li, and Oriol Vinyals. 2018. Representation learning with contrastive predictive coding. arXiv preprint arXiv:1807.03748 (2018)."},{"key":"e_1_3_2_1_33_1","volume-title":"Auxiliary Cross-Modal Representation Learning with Triplet Loss Functions for Online Handwriting Recognition. arXiv preprint arXiv:2202.07901","author":"Ott Felix","year":"2022","unstructured":"Felix Ott, David R\u00fcgamer, Lucas Heublein, Bernd Bischl, and Christopher Mutschler. 2022. Auxiliary Cross-Modal Representation Learning with Triplet Loss Functions for Online Handwriting Recognition. arXiv preprint arXiv:2202.07901 (2022)."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413686"},{"key":"e_1_3_2_1_35_1","unstructured":"Mandela Patrick Yuki Asano Polina Kuznetsova Ruth Fong Joao F Henriques Geoffrey Zweig and Andrea Vedaldi. 2020. Multi-modal self-supervision from generalized data transformations. (2020)."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00021"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053895"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV45572.2020.9093616"},{"volume-title":"Hear the Pixels. In 2020 IEEE Winter Conference on Applications of Computer Vision (WACV).","author":"Ramaswamy J.","key":"e_1_3_2_1_39_1","unstructured":"J. Ramaswamy and S. Das. 2020. See the Sound, Hear the Pixels. In 2020 IEEE Winter Conference on Applications of Computer Vision (WACV)."},{"key":"e_1_3_2_1_40_1","unstructured":"Ruslan Salakhutdinov and Geoff Hinton. 2007. Learning a nonlinear embedding by preserving class neighbourhood structure. In Artificial intelligence and statistics. PMLR 412--419."},{"key":"e_1_3_2_1_41_1","volume-title":"Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman. 2014. Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00678"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58621-8_45"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00493"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01216-8_16"},{"key":"e_1_3_2_1_46_1","volume-title":"Multimodal self-supervised learning of general audio representations. arXiv preprint arXiv:2104.12807","author":"Wang Luyu","year":"2021","unstructured":"Luyu Wang, Pauline Luc, Adria Recasens, Jean-Baptiste Alayrac, and Aaron van den Oord. 2021. Multimodal self-supervised learning of general audio representations. arXiv preprint arXiv:2104.12807 (2021)."},{"key":"e_1_3_2_1_47_1","volume-title":"Listen and Pay More Attention: Fusing Multi-Modal Information for Video Violence Detection. In ICASSP 2022--2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE","author":"Wei Dong-Lai","year":"2022","unstructured":"Dong-Lai Wei, Chen-Geng Liu, Yang Liu, Jing Liu, Xiao-Guang Zhu, and Xin- Hua Zeng. 2022. Look, Listen and Pay More Attention: Fusing Multi-Modal Information for Video Violence Detection. In ICASSP 2022--2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 1980--1984."},{"key":"e_1_3_2_1_48_1","volume-title":"Learning in audio-visual context: A review, analysis, and new perspective. arXiv preprint arXiv:2208.09579","author":"Wei Yake","year":"2022","unstructured":"Yake Wei, Di Hu, Yapeng Tian, and Xuelong Li. 2022. Learning in audio-visual context: A review, analysis, and new perspective. arXiv preprint arXiv:2208.09579 (2022)."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2021.3062192"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00639"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01234-2_42"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00393"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01936"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413581"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"crossref","unstructured":"H. Xuan Z. Zhang S. Chen J. Yang and Y. Yan. 2020. Cross-Modal Attention Network for Temporal Inconsistent Audio-Visual Event Localization. In Association for the Advancement of Artificial Intelligence (AAAI). 279--286.","DOI":"10.1609\/aaai.v34i01.5361"},{"key":"e_1_3_2_1_56_1","volume-title":"MCL: A Contrastive Learning Method for Multimodal Data Fusion in Violence Detection","author":"Yang Liu","year":"2022","unstructured":"Liu Yang, Zhenjie Wu, Junkun Hong, and Jun Long. 2022. MCL: A Contrastive Learning Method for Multimodal Data Fusion in Violence Detection. IEEE Signal Processing Letters (2022)."},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547868"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1145\/3387164"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00833"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00833"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00610"}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Ottawa ON Canada","acronym":"MM '23"},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3613805","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3613805","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T23:59:55Z","timestamp":1755820795000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3613805"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":61,"alternative-id":["10.1145\/3581783.3613805","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3613805","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}