{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T17:13:35Z","timestamp":1783790015009,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":68,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,10,10]],"date-time":"2022-10-10T00:00:00Z","timestamp":1665360000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Natural Science Foundation of China","award":["No. 62172101?No. 61976057"],"award-info":[{"award-number":["No. 62172101?No. 61976057"]}]},{"name":"SPMI Innovation and Technology Fund Projects","award":["SAST2020-110"],"award-info":[{"award-number":["SAST2020-110"]}]},{"DOI":"10.13039\/501100003399","name":"Science and Technology Commission of Shanghai Municipality","doi-asserted-by":"publisher","award":["No. 21511101000, No. 21511100602"],"award-info":[{"award-number":["No. 21511101000, No. 21511100602"]}],"id":[{"id":"10.13039\/501100003399","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,10,10]]},"DOI":"10.1145\/3503161.3547868","type":"proceedings-article","created":{"date-parts":[[2022,10,10]],"date-time":"2022-10-10T15:42:35Z","timestamp":1665416555000},"page":"6278-6287","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":56,"title":["Modality-aware Contrastive Instance Learning with Self-Distillation for Weakly-Supervised Audio-Visual Violence Detection"],"prefix":"10.1145","author":[{"given":"Jiashuo","family":"Yu","sequence":"first","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jinyu","family":"Liu","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ying","family":"Cheng","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rui","family":"Feng","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuejie","family":"Zhang","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2022,10,10]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"crossref","unstructured":"Relja Arandjelovic and Andrew Zisserman. 2017. Look listen and learn. In ICCV. 609--617.  Relja Arandjelovic and Andrew Zisserman. 2017. Look listen and learn. In ICCV. 609--617.","DOI":"10.1109\/ICCV.2017.73"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"crossref","unstructured":"Relja Arandjelovic and Andrew Zisserman. 2018. Objects that sound. In ECCV. 435--451.  Relja Arandjelovic and Andrew Zisserman. 2018. Objects that sound. In ECCV. 435--451.","DOI":"10.1007\/978-3-030-01246-5_27"},{"key":"e_1_3_2_2_3_1","volume-title":"Jamie Ryan Kiros, and Geoffrey E Hinton","author":"Ba Jimmy Lei","year":"2016","unstructured":"Jimmy Lei Ba , Jamie Ryan Kiros, and Geoffrey E Hinton . 2016 . Layer normalization. arXiv preprint arXiv:1607.06450. Jimmy Lei Ba, Jamie Ryan Kiros, and Geoffrey E Hinton. 2016. Layer normalization. arXiv preprint arXiv:1607.06450."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.5555\/2044575.2044624"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/1150402.1150464"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01603"},{"key":"e_1_3_2_2_9_1","volume-title":"Counterfactual samples synthesizing and training for robust visual question answering. arXiv preprint arXiv:2110.01013","author":"Chen Long","year":"2021","unstructured":"Long Chen , Yuhang Zheng , Yulei Niu , Hanwang Zhang , and Jun Xiao . 2021. Counterfactual samples synthesizing and training for robust visual question answering. arXiv preprint arXiv:2110.01013 ( 2021 ). Long Chen, Yuhang Zheng, Yulei Niu, Hanwang Zhang, and Jun Xiao. 2021. Counterfactual samples synthesizing and training for robust visual question answering. arXiv preprint arXiv:2110.01013 (2021)."},{"key":"e_1_3_2_2_10_1","volume-title":"International conference on machine learning. PMLR, 1597--1607","author":"Chen Ting","year":"2020","unstructured":"Ting Chen , Simon Kornblith , Mohammad Norouzi , and Geoffrey Hinton . 2020 . A simple framework for contrastive learning of visual representations . In International conference on machine learning. PMLR, 1597--1607 . Ting Chen, Simon Kornblith, Mohammad Norouzi, and Geoffrey Hinton. 2020. A simple framework for contrastive learning of visual representations. In International conference on machine learning. PMLR, 1597--1607."},{"key":"e_1_3_2_2_11_1","volume-title":"Big self-supervised models are strong semi-supervised learners. Advances in neural information processing systems 33","author":"Chen Ting","year":"2020","unstructured":"Ting Chen , Simon Kornblith , Kevin Swersky , Mohammad Norouzi , and Geoffrey E Hinton . 2020. Big self-supervised models are strong semi-supervised learners. Advances in neural information processing systems 33 ( 2020 ), 22243--22255. Ting Chen, Simon Kornblith, Kevin Swersky, Mohammad Norouzi, and Geoffrey E Hinton. 2020. Big self-supervised models are strong semi-supervised learners. Advances in neural information processing systems 33 (2020), 22243--22255."},{"key":"e_1_3_2_2_12_1","volume-title":"CIL: Contrastive Instance Learning Framework for Distantly Supervised Relation Extraction. arXiv preprint arXiv:2106.10855.","author":"Chen Tao","year":"2021","unstructured":"Tao Chen , Haizhou Shi , Siliang Tang , Zhigang Chen , Fei Wu , and Yueting Zhuang . 2021 . CIL: Contrastive Instance Learning Framework for Distantly Supervised Relation Extraction. arXiv preprint arXiv:2106.10855. Tao Chen, Haizhou Shi, Siliang Tang, Zhigang Chen, Fei Wu, and Yueting Zhuang. 2021. CIL: Contrastive Instance Learning Framework for Distantly Supervised Relation Extraction. arXiv preprint arXiv:2106.10855."},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00694"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"crossref","unstructured":"Ying Cheng Ruize Wang Zhihao Pan Rui Feng and Yuejie Zhang. 2020. Look listen and attend: Co-attention network for self-supervised audio-visual representation learning. In ACM MM. 3884--3892.  Ying Cheng Ruize Wang Zhihao Pan Rui Feng and Yuejie Zhang. 2020. Look listen and attend: Co-attention network for self-supervised audio-visual representation learning. In ACM MM. 3884--3892.","DOI":"10.1145\/3394171.3413869"},{"key":"e_1_3_2_2_15_1","volume-title":"Electra: Pre-training text encoders as discriminators rather than generators. arXiv preprint arXiv:2003.10555","author":"Clark Kevin","year":"2020","unstructured":"Kevin Clark , Minh-Thang Luong , Quoc V Le , and Christopher D Manning . 2020 . Electra: Pre-training text encoders as discriminators rather than generators. arXiv preprint arXiv:2003.10555 (2020). Kevin Clark, Minh-Thang Luong, Quoc V Le, and Christopher D Manning. 2020. Electra: Pre-training text encoders as discriminators rather than generators. arXiv preprint arXiv:2003.10555 (2020)."},{"key":"e_1_3_2_2_16_1","volume-title":"Contrastive learning for image captioning. Advances in Neural Information Processing Systems 30","author":"Dai Bo","year":"2017","unstructured":"Bo Dai and Dahua Lin . 2017. Contrastive learning for image captioning. Advances in Neural Information Processing Systems 30 ( 2017 ). Bo Dai and Dahua Lin. 2017. Contrastive learning for image captioning. Advances in Neural Information Processing Systems 30 (2017)."},{"key":"e_1_3_2_2_17_1","volume-title":"CONTaiNER: Few-Shot Named Entity Recognition via Contrastive Learning. arXiv preprint arXiv:2109.07589","author":"Sarathi Das Sarkar Snigdha","year":"2021","unstructured":"Sarkar Snigdha Sarathi Das , Arzoo Katiyar , Rebecca J Passonneau , and Rui Zhang . 2021. CONTaiNER: Few-Shot Named Entity Recognition via Contrastive Learning. arXiv preprint arXiv:2109.07589 ( 2021 ). Sarkar Snigdha Sarathi Das, Arzoo Katiyar, Rebecca J Passonneau, and Rui Zhang. 2021. CONTaiNER: Few-Shot Named Entity Recognition via Contrastive Learning. arXiv preprint arXiv:2109.07589 (2021)."},{"key":"e_1_3_2_2_18_1","volume-title":"2014 international conference on computer vision theory and applications (VISAPP)","volume":"2","author":"Deniz Oscar","year":"2014","unstructured":"Oscar Deniz , Ismael Serrano , Gloria Bueno , and Tae-Kyun Kim . 2014 . Fast violence detection in video . In 2014 international conference on computer vision theory and applications (VISAPP) , Vol. 2 . IEEE, 478--485. Oscar Deniz, Ismael Serrano, Gloria Bueno, and Tae-Kyun Kim. 2014. Fast violence detection in video. In 2014 international conference on computer vision theory and applications (VISAPP), Vol. 2. IEEE, 478--485."},{"key":"e_1_3_2_2_19_1","volume-title":"Seed: Self-supervised distillation for visual representation. arXiv preprint arXiv:2101.04731.","author":"Fang Zhiyuan","year":"2021","unstructured":"Zhiyuan Fang , Jianfeng Wang , Lijuan Wang , Lei Zhang , Yezhou Yang , and Zicheng Liu . 2021 . Seed: Self-supervised distillation for visual representation. arXiv preprint arXiv:2101.04731. Zhiyuan Fang, Jianfeng Wang, Lijuan Wang, Lei Zhang, Yezhou Yang, and Zicheng Liu. 2021. Seed: Self-supervised distillation for visual representation. arXiv preprint arXiv:2101.04731."},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01379"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01237-3_7"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"e_1_3_2_2_23_1","first-page":"21271","article-title":"Bootstrap your own latent-a new approach to self-supervised learning","volume":"33","author":"Grill Jean-Bastien","year":"2020","unstructured":"Jean-Bastien Grill , Florian Strub , Florent Altch\u00e9 , Corentin Tallec , Pierre Richemond , Elena Buchatskaya , Carl Doersch , Bernardo Avila Pires , Zhaohan Guo , Mohammad Gheshlaghi Azar , 2020 . Bootstrap your own latent-a new approach to self-supervised learning . Advances in Neural Information Processing Systems 33 (2020), 21271 -- 21284 . Jean-Bastien Grill, Florian Strub, Florent Altch\u00e9, Corentin Tallec, Pierre Richemond, Elena Buchatskaya, Carl Doersch, Bernardo Avila Pires, Zhaohan Guo, Mohammad Gheshlaghi Azar, et al . 2020. Bootstrap your own latent-a new approach to self-supervised learning. Advances in Neural Information Processing Systems 33 (2020), 21271--21284.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.86"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"e_1_3_2_2_26_1","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep residual learning for image recognition. In CVPR. 770--778.  Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep residual learning for image recognition. In CVPR. 770--778."},{"key":"e_1_3_2_2_27_1","volume-title":"Jort F Gemmeke, Aren Jansen, R Channing Moore, Manoj Plakal, Devin Platt, Rif A Saurous, Bryan Seybold, et al.","author":"Hershey Shawn","year":"2017","unstructured":"Shawn Hershey , Sourish Chaudhuri , Daniel PW Ellis , Jort F Gemmeke, Aren Jansen, R Channing Moore, Manoj Plakal, Devin Platt, Rif A Saurous, Bryan Seybold, et al. 2017 . CNN architectures for large-scale audio classification. In ICASSP. 131--135. Shawn Hershey, Sourish Chaudhuri, Daniel PW Ellis, Jort F Gemmeke, Aren Jansen, R Channing Moore, Manoj Plakal, Devin Platt, Rif A Saurous, Bryan Seybold, et al. 2017. CNN architectures for large-scale audio classification. In ICASSP. 131--135."},{"key":"e_1_3_2_2_28_1","volume-title":"et al","author":"Hinton Geoffrey","year":"2015","unstructured":"Geoffrey Hinton , Oriol Vinyals , Jeff Dean , et al . 2015 . Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531 2, 7 (2015). Geoffrey Hinton, Oriol Vinyals, Jeff Dean, et al . 2015. Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531 2, 7 (2015)."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.96"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.3390\/app9224963"},{"key":"e_1_3_2_2_31_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980","author":"Kingma Diederik P","year":"2014","unstructured":"Diederik P Kingma and Jimmy Ba . 2014 . Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014). Diederik P Kingma and Jimmy Ba. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)."},{"key":"e_1_3_2_2_32_1","unstructured":"Bruno Korbar Du Tran and Lorenzo Torresani. 2018. Cooperative Learning of Audio and Video Models from Self-Supervised Synchronization. In NeurIPS. 7774--7785.  Bruno Korbar Du Tran and Lorenzo Torresani. 2018. Cooperative Learning of Audio and Video Models from Self-Supervised Synchronization. In NeurIPS. 7774--7785."},{"key":"e_1_3_2_2_33_1","unstructured":"Shuo Li Fang Liu and Licheng Jiao. 2022. Self-Training Multi-Sequence Learning with Transformer for Weakly Supervised Video Anomaly Detection. (2022).  Shuo Li Fang Liu and Licheng Jiao. 2022. Self-Training Multi-Sequence Learning with Transformer for Weakly Supervised Video Anomaly Detection. (2022)."},{"key":"e_1_3_2_2_34_1","volume-title":"Unimo: Towards unified-modal understanding and generation via cross-modal contrastive learning. arXiv preprint arXiv:2012.15409","author":"Li Wei","year":"2020","unstructured":"Wei Li , Can Gao , Guocheng Niu , Xinyan Xiao , Hao Liu , Jiachen Liu , Hua Wu , and Haifeng Wang . 2020 . Unimo: Towards unified-modal understanding and generation via cross-modal contrastive learning. arXiv preprint arXiv:2012.15409 (2020). Wei Li, Can Gao, Guocheng Niu, Xinyan Xiao, Hao Liu, Jiachen Liu, Hua Wu, and Haifeng Wang. 2020. Unimo: Towards unified-modal understanding and generation via cross-modal contrastive learning. arXiv preprint arXiv:2012.15409 (2020)."},{"key":"e_1_3_2_2_35_1","volume-title":"Self-Supervised Video Representation Learning with Motion-Contrastive Perception. arXiv preprint arXiv:2204.04607","author":"Liu Jinyu","year":"2022","unstructured":"Jinyu Liu , Ying Cheng , Yuejie Zhang , Rui-Wei Zhao , and Rui Feng . 2022. Self-Supervised Video Representation Learning with Motion-Contrastive Perception. arXiv preprint arXiv:2204.04607 ( 2022 ). Jinyu Liu, Ying Cheng, Yuejie Zhang, Rui-Wei Zhao, and Rui Feng. 2022. Self-Supervised Video Representation Learning with Motion-Contrastive Perception. arXiv preprint arXiv:2204.04607 (2022)."},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00684"},{"key":"e_1_3_2_2_37_1","unstructured":"Shuang Ma Zhaoyang Zeng Daniel McDuff and Yale Song. 2021. Active Contrastive Learning of Audio-Visual Video Representations. In ICLR. https:\/\/openreview.net\/forum?id=OMizHuea_HB  Shuang Ma Zhaoyang Zeng Daniel McDuff and Yale Song. 2021. Active Contrastive Learning of Audio-Visual Video Representations. In ICLR. https:\/\/openreview.net\/forum?id=OMizHuea_HB"},{"key":"e_1_3_2_2_38_1","unstructured":"Oded Maron and Tom\u00e1s Lozano-P\u00e9rez. 1997. A framework for multiple-instance learning. Advances in neural information processing systems 10.  Oded Maron and Tom\u00e1s Lozano-P\u00e9rez. 1997. A framework for multiple-instance learning. Advances in neural information processing systems 10."},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01274"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01229"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00706"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"crossref","unstructured":"Andrew Owens and Alexei A Efros. 2018. Audio-visual scene analysis with self-supervised multisensory features. In ECCV. 631--648.  Andrew Owens and Alexei A Efros. 2018. Audio-visual scene analysis with self-supervised multisensory features. In ECCV. 631--648.","DOI":"10.1007\/978-3-030-01231-1_39"},{"key":"e_1_3_2_2_43_1","volume-title":"Violence Detection in Videos Based on Fusing Visual and Audio Information. In ICASSP 2021--2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2260--2264","author":"Pang Wen-Feng","year":"2021","unstructured":"Wen-Feng Pang , Qian-Hua He , Yong-jian Hu, and Yan-Xiong Li . 2021 . Violence Detection in Videos Based on Fusing Visual and Audio Information. In ICASSP 2021--2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2260--2264 . Wen-Feng Pang, Qian-Hua He, Yong-jian Hu, and Yan-Xiong Li. 2021. Violence Detection in Videos Based on Fusing Visual and Audio Information. In ICASSP 2021--2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2260--2264."},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054018"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682833"},{"key":"e_1_3_2_2_46_1","volume-title":"Learning from context or names? an empirical study on neural relation extraction. arXiv preprint arXiv:2010.01923","author":"Peng Hao","year":"2020","unstructured":"Hao Peng , Tianyu Gao , Xu Han , Yankai Lin , Peng Li , Zhiyuan Liu , Maosong Sun , and Jie Zhou . 2020. Learning from context or names? an empirical study on neural relation extraction. arXiv preprint arXiv:2010.01923 ( 2020 ). Hao Peng, Tianyu Gao, Xu Han, Yankai Lin, Peng Li, Zhiyuan Liu, Maosong Sun, and Jie Zhou. 2020. Learning from context or names? an empirical study on neural relation extraction. arXiv preprint arXiv:2010.01923 (2020)."},{"key":"e_1_3_2_2_47_1","volume-title":"Kamal Nasrollahi, Fahad Shahbaz Khan, Thomas B Moeslund, and Mubarak Shah.","author":"Ristea Nicolae-Catalin","year":"2021","unstructured":"Nicolae-Catalin Ristea , Neelu Madan , Radu Tudor Ionescu , Kamal Nasrollahi, Fahad Shahbaz Khan, Thomas B Moeslund, and Mubarak Shah. 2021 . Self- Supervised Predictive Convolutional Attentive Block for Anomaly Detection . arXiv preprint arXiv:2111.09099. Nicolae-Catalin Ristea, Neelu Madan, Radu Tudor Ionescu, Kamal Nasrollahi, Fahad Shahbaz Khan, Thomas B Moeslund, and Mubarak Shah. 2021. Self- Supervised Predictive Convolutional Attentive Block for Anomaly Detection. arXiv preprint arXiv:2111.09099."},{"key":"e_1_3_2_2_48_1","volume-title":"Support vector method for novelty detection. Advances in neural information processing systems 12","author":"Sch\u00f6lkopf Bernhard","year":"1999","unstructured":"Bernhard Sch\u00f6lkopf , Robert C Williamson , Alex Smola , John Shawe-Taylor , and John Platt . 1999. Support vector method for novelty detection. Advances in neural information processing systems 12 ( 1999 ). Bernhard Sch\u00f6lkopf, Robert C Williamson, Alex Smola, John Shawe-Taylor, and John Platt. 1999. Support vector method for novelty detection. Advances in neural information processing systems 12 (1999)."},{"key":"e_1_3_2_2_49_1","volume-title":"Contrastive visual-linguistic pretraining. arXiv preprint arXiv:2007.13135","author":"Shi Lei","year":"2020","unstructured":"Lei Shi , Kai Shuang , Shijie Geng , Peng Su , Zhengkai Jiang , Peng Gao , Zuohui Fu , Gerard de Melo , and Sen Su. 2020. Contrastive visual-linguistic pretraining. arXiv preprint arXiv:2007.13135 ( 2020 ). Lei Shi, Kai Shuang, Shijie Geng, Peng Su, Zhengkai Jiang, Peng Gao, Zuohui Fu, Gerard de Melo, and Sen Su. 2020. Contrastive visual-linguistic pretraining. arXiv preprint arXiv:2007.13135 (2020)."},{"key":"e_1_3_2_2_50_1","volume-title":"TaCL: Improving BERT Pre-training with Token-aware Contrastive Learning. arXiv preprint arXiv:2111.04198","author":"Su Yixuan","year":"2021","unstructured":"Yixuan Su , Fangyu Liu , Zaiqiao Meng , Lei Shu , Ehsan Shareghi , and Nigel Collier . 2021. TaCL: Improving BERT Pre-training with Token-aware Contrastive Learning. arXiv preprint arXiv:2111.04198 ( 2021 ). Yixuan Su, Fangyu Liu, Zaiqiao Meng, Lei Shu, Ehsan Shareghi, and Nigel Collier. 2021. TaCL: Improving BERT Pre-training with Token-aware Contrastive Learning. arXiv preprint arXiv:2111.04198 (2021)."},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00678"},{"key":"e_1_3_2_2_52_1","unstructured":"Yonglong Tian Dilip Krishnan and Phillip Isola. 2019. Contrastive representation distillation. arXiv preprint arXiv:1910.10699.  Yonglong Tian Dilip Krishnan and Phillip Isola. 2019. Contrastive representation distillation. arXiv preprint arXiv:1910.10699."},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58580-8_26"},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00493"},{"key":"e_1_3_2_2_55_1","unstructured":"Aaron Van den Oord Yazhe Li Oriol Vinyals etal 2018. Representation learning with contrastive predictive coding. arXiv preprint arXiv:1807.03748 2 3 (2018) 4.  Aaron Van den Oord Yazhe Li Oriol Vinyals et al. 2018. Representation learning with contrastive predictive coding. arXiv preprint arXiv:1807.03748 2 3 (2018) 4."},{"key":"e_1_3_2_2_56_1","article-title":"Visualizing data using t-SNE","volume":"9","author":"der Maaten Laurens Van","year":"2008","unstructured":"Laurens Van der Maaten and Geoffrey Hinton . 2008 . Visualizing data using t-SNE . Journal of machine learning research 9 , 11 (2008). Laurens Van der Maaten and Geoffrey Hinton. 2008. Visualizing data using t-SNE. Journal of machine learning research 9, 11 (2008).","journal-title":"Journal of machine learning research"},{"key":"e_1_3_2_2_57_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez Lukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30.  Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez Lukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30."},{"key":"e_1_3_2_2_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3159811"},{"key":"e_1_3_2_2_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00221"},{"key":"e_1_3_2_2_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00558"},{"key":"e_1_3_2_2_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2021.3062192"},{"key":"e_1_3_2_2_62_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58577-8_20"},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00393"},{"key":"e_1_3_2_2_64_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01070"},{"key":"e_1_3_2_2_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00089"},{"key":"e_1_3_2_2_66_1","volume-title":"Vision-Language Pre-Training with Triple Contrastive Learning. arXiv preprint arXiv:2202.10401","author":"Yang Jinyu","year":"2022","unstructured":"Jinyu Yang , Jiali Duan , Son Tran , Yi Xu , Sampath Chanda , Liqun Chen , Belinda Zeng , Trishul Chilimbi , and Junzhou Huang . 2022. Vision-Language Pre-Training with Triple Contrastive Learning. arXiv preprint arXiv:2202.10401 ( 2022 ). Jinyu Yang, Jiali Duan, Son Tran, Yi Xu, Sampath Chanda, Liqun Chen, Belinda Zeng, Trishul Chilimbi, and Junzhou Huang. 2022. Vision-Language Pre-Training with Triple Contrastive Learning. arXiv preprint arXiv:2202.10401 (2022)."},{"key":"e_1_3_2_2_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP.2019.8803657"},{"key":"e_1_3_2_2_68_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-015-2648-8"}],"event":{"name":"MM '22: The 30th ACM International Conference on Multimedia","location":"Lisboa Portugal","acronym":"MM '22","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 30th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503161.3547868","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3503161.3547868","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:02:35Z","timestamp":1750186955000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503161.3547868"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,10]]},"references-count":68,"alternative-id":["10.1145\/3503161.3547868","10.1145\/3503161"],"URL":"https:\/\/doi.org\/10.1145\/3503161.3547868","relation":{},"subject":[],"published":{"date-parts":[[2022,10,10]]},"assertion":[{"value":"2022-10-10","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}