{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T05:04:16Z","timestamp":1750309456156,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":38,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,12,3]],"date-time":"2024-12-03T00:00:00Z","timestamp":1733184000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,12,3]]},"DOI":"10.1145\/3696409.3700225","type":"proceedings-article","created":{"date-parts":[[2024,12,28]],"date-time":"2024-12-28T09:55:23Z","timestamp":1735379723000},"page":"1-8","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Enhancing Modality Representation and Alignment for Multimodal Cold-start Active Learning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2502-3500","authenticated-orcid":false,"given":"Meng","family":"Shen","sequence":"first","affiliation":[{"name":"Nanyang Technological University, Singapore, SG"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-8779-0637","authenticated-orcid":false,"given":"Yake","family":"Wei","sequence":"additional","affiliation":[{"name":"Renmin University of China, Beijing, CN"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4686-2768","authenticated-orcid":false,"given":"Jianxiong (Terry)","family":"Yin","sequence":"additional","affiliation":[{"name":"NVIDIA AI Tech Centre, Singapore, SG"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7788-8368","authenticated-orcid":false,"given":"Deepu","family":"Rajan","sequence":"additional","affiliation":[{"name":"Nanyang Technological University, Singapore, SG"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7118-6733","authenticated-orcid":false,"given":"Di","family":"Hu","sequence":"additional","affiliation":[{"name":"Renmin University of China, Beijing, CN"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4958-9237","authenticated-orcid":false,"given":"Simon","family":"See","sequence":"additional","affiliation":[{"name":"NVIDIA AI Tech Centre, Singapore, SG"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,12,28]]},"reference":[{"key":"e_1_3_3_1_2_2","volume-title":"Advances in Neural Information Processing Systems 33: Annual Conference on Neural Information Processing Systems 2020, NeurIPS 2020, December 6-12, 2020, virtual","author":"Alwassel Humam","year":"2020","unstructured":"Humam Alwassel, Dhruv Mahajan, Bruno Korbar, Lorenzo Torresani, Bernard Ghanem, and Du Tran. 2020. Self-Supervised Learning by Cross-Modal Audio-Video Clustering. In Advances in Neural Information Processing Systems 33: Annual Conference on Neural Information Processing Systems 2020, NeurIPS 2020, December 6-12, 2020, virtual. https:\/\/proceedings.neurips.cc\/paper\/2020\/hash\/6f2268bd1d3d3ebaabb04d6b5d099425-Abstract.html"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.73"},{"key":"e_1_3_3_1_4_2","volume-title":"8th International Conference on Learning Representations, ICLR 2020, Addis Ababa, Ethiopia, April 26-30, 2020","author":"Ash Jordan\u00a0T.","year":"2020","unstructured":"Jordan\u00a0T. Ash, Chicheng Zhang, Akshay Krishnamurthy, John Langford, and Alekh Agarwal. 2020. Deep Batch Active Learning by Diverse, Uncertain Gradient Lower Bounds. In 8th International Conference on Learning Representations, ICLR 2020, Addis Ababa, Ethiopia, April 26-30, 2020. OpenReview.net. https:\/\/openreview.net\/forum?id=ryghZJBKPS"},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"publisher","unstructured":"Tadas Baltrusaitis Chaitanya Ahuja and Louis-Philippe Morency. 2019. Multimodal Machine Learning: A Survey and Taxonomy. IEEE Trans. Pattern Anal. Mach. Intell. 41 2 (2019) 423\u2013443. 10.1109\/TPAMI.2018.2798607 https:\/\/dl.acm.org\/doi\/10.1109\/TPAMI.2018.2798607","DOI":"10.1109\/TPAMI.2018.2798607"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00946"},{"key":"e_1_3_3_1_7_2","volume-title":"Advances in Neural Information Processing Systems 33: Annual Conference on Neural Information Processing Systems 2020, NeurIPS 2020, December 6-12, 2020, virtual","author":"Caron Mathilde","year":"2020","unstructured":"Mathilde Caron, Ishan Misra, Julien Mairal, Priya Goyal, Piotr Bojanowski, and Armand Joulin. 2020. Unsupervised Learning of Visual Features by Contrasting Cluster Assignments. In Advances in Neural Information Processing Systems 33: Annual Conference on Neural Information Processing Systems 2020, NeurIPS 2020, December 6-12, 2020, virtual. https:\/\/proceedings.neurips.cc\/paper\/2020\/hash\/70feb62b69f16e0238f741fab228fec2-Abstract.html"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053174"},{"key":"e_1_3_3_1_10_2","series-title":"Proceedings of Machine Learning Research","first-page":"496","volume-title":"Medical Imaging with Deep Learning, MIDL 2023, 10-12 July 2023, Nashville, TN, USA","volume":"227","author":"Chen Liangyu","year":"2023","unstructured":"Liangyu Chen, Yutong Bai, Siyu Huang, Yongyi Lu, Bihan Wen, Alan\u00a0L. Yuille, and Zongwei Zhou. 2023. Making Your First Choice: To Address Cold Start Problem in Medical Active Learning. In Medical Imaging with Deep Learning, MIDL 2023, 10-12 July 2023, Nashville, TN, USA(Proceedings of Machine Learning Research, Vol.\u00a0227). PMLR, 496\u2013525. https:\/\/proceedings.mlr.press\/v227\/chen24a.html"},{"key":"e_1_3_3_1_11_2","unstructured":"Xinlei Chen Haoqi Fan Ross\u00a0B. Girshick and Kaiming He. 2020. Improved Baselines with Momentum Contrastive Learning. CoRR abs\/2003.04297 (2020). arXiv:https:\/\/arXiv.org\/abs\/2003.04297https:\/\/arxiv.org\/abs\/2003.04297"},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"publisher","DOI":"10.18653\/V1\/N19-1423"},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095889"},{"key":"e_1_3_3_1_14_2","series-title":"Proceedings of Machine Learning Research","first-page":"8175","volume-title":"International Conference on Machine Learning, ICML 2022, 17-23 July 2022, Baltimore, Maryland, USA","volume":"162","author":"Hacohen Guy","year":"2022","unstructured":"Guy Hacohen, Avihu Dekel, and Daphna Weinshall. 2022. Active Learning on a Budget: Opposite Strategies Suit High and Low Budgets. In International Conference on Machine Learning, ICML 2022, 17-23 July 2022, Baltimore, Maryland, USA(Proceedings of Machine Learning Research, Vol.\u00a0162). PMLR, 8175\u20138195. https:\/\/proceedings.mlr.press\/v162\/hacohen22a.html"},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i7.25967"},{"key":"e_1_3_3_1_18_2","unstructured":"Will Kay Jo\u00e3o Carreira Karen Simonyan Brian Zhang Chloe Hillier Sudheendra Vijayanarasimhan Fabio Viola Tim Green Trevor Back Paul Natsev Mustafa Suleyman and Andrew Zisserman. 2017. The Kinetics Human Action Video Dataset. CoRR abs\/1705.06950 (2017). arXiv:https:\/\/arXiv.org\/abs\/1705.06950http:\/\/arxiv.org\/abs\/1705.06950"},{"key":"e_1_3_3_1_19_2","volume-title":"9th International Conference on Learning Representations, ICLR 2021, Virtual Event, Austria, May 3-7, 2021","author":"Li Junnan","year":"2021","unstructured":"Junnan Li, Pan Zhou, Caiming Xiong, and Steven C.\u00a0H. Hoi. 2021. Prototypical Contrastive Learning of Unsupervised Representations. In 9th International Conference on Learning Representations, ICLR 2021, Virtual Event, Austria, May 3-7, 2021. OpenReview.net. https:\/\/openreview.net\/forum?id=KmykpuSrjcq"},{"key":"e_1_3_3_1_20_2","volume-title":"Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022, NeurIPS 2022, New Orleans, LA, USA, November 28 - December 9, 2022","author":"Liang Weixin","year":"2022","unstructured":"Weixin Liang, Yuhui Zhang, Yongchan Kwon, Serena Yeung, and James\u00a0Y. Zou. 2022. Mind the Gap: Understanding the Modality Gap in Multi-modal Contrastive Representation Learning. In Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022, NeurIPS 2022, New Orleans, LA, USA, November 28 - December 9, 2022. http:\/\/papers.nips.cc\/paper_files\/paper\/2022\/hash\/702f4db7543a7432431df588d57bc7c9-Abstract-Conference.html"},{"key":"e_1_3_3_1_21_2","volume-title":"7th International Conference on Learning Representations, ICLR 2019, New Orleans, LA, USA, May 6-9, 2019","author":"Loshchilov Ilya","year":"2019","unstructured":"Ilya Loshchilov and Frank Hutter. 2019. Decoupled Weight Decay Regularization. In 7th International Conference on Learning Representations, ICLR 2019, New Orleans, LA, USA, May 6-9, 2019. OpenReview.net. https:\/\/openreview.net\/forum?id=Bkg6RiCqY7"},{"key":"e_1_3_3_1_22_2","volume-title":"9th International Conference on Learning Representations, ICLR 2021, Virtual Event, Austria, May 3-7, 2021","author":"Ma Shuang","year":"2021","unstructured":"Shuang Ma, Zhaoyang Zeng, Daniel McDuff, and Yale Song. 2021. Active Contrastive Learning of Audio-Visual Video Representations. In 9th International Conference on Learning Representations, ICLR 2021, Virtual Event, Austria, May 3-7, 2021. OpenReview.net. https:\/\/openreview.net\/forum?id=OMizHuea_HB"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01192"},{"key":"e_1_3_3_1_24_2","series-title":"Proceedings of Machine Learning Research","first-page":"8748","volume-title":"Proceedings of the 38th International Conference on Machine Learning, ICML 2021, 18-24 July 2021, Virtual Event","volume":"139","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In Proceedings of the 38th International Conference on Machine Learning, ICML 2021, 18-24 July 2021, Virtual Event(Proceedings of Machine Learning Research, Vol.\u00a0139). PMLR, 8748\u20138763. http:\/\/proceedings.mlr.press\/v139\/radford21a.html"},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","unstructured":"Pengzhen Ren Yun Xiao Xiaojun Chang Po-Yao Huang Zhihui Li Brij\u00a0B. Gupta Xiaojiang Chen and Xin Wang. 2022. A Survey of Deep Active Learning. ACM Comput. Surv. 54 9 (2022) 180:1\u2013180:40. 10.1145\/3472291https:\/\/dl.acm.org\/doi\/10.1145\/3472291","DOI":"10.1145\/3472291"},{"key":"e_1_3_3_1_26_2","volume-title":"6th International Conference on Learning Representations, ICLR 2018, Vancouver, BC, Canada, April 30 - May 3, 2018, Conference Track Proceedings","author":"Sener Ozan","year":"2018","unstructured":"Ozan Sener and Silvio Savarese. 2018. Active Learning for Convolutional Neural Networks: A Core-Set Approach. In 6th International Conference on Learning Representations, ICLR 2018, Vancouver, BC, Canada, April 30 - May 3, 2018, Conference Track Proceedings. OpenReview.net. https:\/\/openreview.net\/forum?id=H1aIuk-RW"},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612463"},{"key":"e_1_3_3_1_28_2","volume-title":"The Tenth International Conference on Learning Representations, ICLR 2022, Virtual Event, April 25-29, 2022","author":"Shi Bowen","year":"2022","unstructured":"Bowen Shi, Wei-Ning Hsu, Kushal Lakhotia, and Abdelrahman Mohamed. 2022. Learning Audio-Visual Speech Representation by Masked Multimodal Cluster Prediction. In The Tenth International Conference on Learning Representations, ICLR 2022, Virtual Event, April 25-29, 2022. OpenReview.net. https:\/\/openreview.net\/forum?id=Z1Qlm11uOM"},{"key":"e_1_3_3_1_29_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.746"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00675"},{"key":"e_1_3_3_1_31_2","unstructured":"A\u00e4ron van\u00a0den Oord Yazhe Li and Oriol Vinyals. 2018. Representation Learning with Contrastive Predictive Coding. CoRR abs\/1807.03748 (2018). arXiv:https:\/\/arXiv.org\/abs\/1807.03748http:\/\/arxiv.org\/abs\/1807.03748"},{"key":"e_1_3_3_1_32_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICMEW.2015.7169757"},{"key":"e_1_3_3_1_33_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02271"},{"key":"e_1_3_3_1_34_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.544"},{"key":"e_1_3_3_1_35_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19809-0_34"},{"key":"e_1_3_3_1_36_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.637"},{"key":"e_1_3_3_1_37_2","doi-asserted-by":"publisher","unstructured":"Chao Zhang Zichao Yang Xiaodong He and Li Deng. 2020. Multimodal Intelligence: Representation Learning Information Fusion and Applications. IEEE J. Sel. Top. Signal Process. 14 3 (2020) 478\u2013493. 10.1109\/JSTSP.2020.2987728","DOI":"10.1109\/JSTSP.2020.2987728"},{"key":"e_1_3_3_1_38_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00148"},{"key":"e_1_3_3_1_39_2","doi-asserted-by":"publisher","unstructured":"Yongshuo Zong Oisin\u00a0Mac Aodha and Timothy\u00a0M. Hospedales. 2023. Self-Supervised Multimodal Learning: A Survey. CoRR abs\/2304.01008 (2023). 10.48550\/ARXIV.2304.01008 arXiv:https:\/\/arXiv.org\/abs\/2304.01008","DOI":"10.48550\/ARXIV.2304.01008"}],"event":{"name":"MMAsia '24: ACM Multimedia Asia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Auckland New Zealand","acronym":"MMAsia '24"},"container-title":["Proceedings of the 6th ACM International Conference on Multimedia in Asia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3696409.3700225","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3696409.3700225","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:10:15Z","timestamp":1750295415000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3696409.3700225"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,3]]},"references-count":38,"alternative-id":["10.1145\/3696409.3700225","10.1145\/3696409"],"URL":"https:\/\/doi.org\/10.1145\/3696409.3700225","relation":{},"subject":[],"published":{"date-parts":[[2024,12,3]]},"assertion":[{"value":"2024-12-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}