{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T05:49:44Z","timestamp":1777873784418,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":42,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,8,3]]},"DOI":"10.1145\/3711896.3737195","type":"proceedings-article","created":{"date-parts":[[2025,8,3]],"date-time":"2025-08-03T21:05:41Z","timestamp":1754255141000},"page":"4872-4881","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Audio-Enhanced Vision-Language Modeling with Latent Space Broadening for High Quality Data Expansion"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-4419-7611","authenticated-orcid":false,"given":"Yu","family":"Sun","sequence":"first","affiliation":[{"name":"ByteDance Inc., San Jose, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-4302-0592","authenticated-orcid":false,"given":"Yin","family":"Li","sequence":"additional","affiliation":[{"name":"ByteDance Inc., San Jose, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6905-1242","authenticated-orcid":false,"given":"Ruixiao","family":"Sun","sequence":"additional","affiliation":[{"name":"ByteDance Inc., San Jose, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6707-5187","authenticated-orcid":false,"given":"Chunhui","family":"Liu","sequence":"additional","affiliation":[{"name":"ByteDance Inc., San Jose, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8529-9491","authenticated-orcid":false,"given":"Fangming","family":"Zhou","sequence":"additional","affiliation":[{"name":"ByteDance Inc., Beijing, Beijing, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5882-6580","authenticated-orcid":false,"given":"Ze","family":"Jin","sequence":"additional","affiliation":[{"name":"ByteDance Inc., Bellevue, WA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-1336-2719","authenticated-orcid":false,"given":"Linjie","family":"Wang","sequence":"additional","affiliation":[{"name":"ByteDance Inc., San Jose, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6457-0411","authenticated-orcid":false,"given":"Xiang","family":"Shen","sequence":"additional","affiliation":[{"name":"ByteDance Inc., Bellevue, WA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-7423-3172","authenticated-orcid":false,"given":"Zhuolin","family":"Hao","sequence":"additional","affiliation":[{"name":"ByteDance Inc., Beijing, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6699-5192","authenticated-orcid":false,"given":"Hongyu","family":"Xiong","sequence":"additional","affiliation":[{"name":"ByteDance Inc., San Jose, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,8,3]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Uncertain Gradient Lower Bounds. In International Conference on Learning Representations.","author":"Ash Jordan T.","year":"2020","unstructured":"Jordan T. Ash, Chicheng Zhang, Akshay Krishnamurthy, John Langford, and Alekh Agarwal. 2020. Deep Batch Active Learning by Diverse, Uncertain Gradient Lower Bounds. In International Conference on Learning Representations."},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/2591796.2591839"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jcss.2008.07.003"},{"key":"e_1_3_2_2_4_1","unstructured":"Hangbo Bao Li Dong Songhao Piao and Furu Wei. 2022. BEiT: BERT Pre-Training of Image Transformers. arXiv:2106.08254 [cs.CV] https:\/\/arxiv.org\/abs\/2106.08254"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/1553374.1553381"},{"key":"e_1_3_2_2_6_1","first-page":"211","volume-title":"Customer Lookalike Modeling: A Study of Machine Learning Techniques for Customer Lookalike Modeling. In Intelligent Data Communication Technologies and Internet of Things: Proceedings of ICICI","author":"Chacko Anna Mariam","year":"2021","unstructured":"Anna Mariam Chacko, Bhuvanapalli Aditya Pranav, Bommanapalli Vijaya Madhvesh, and AS Poornima. 2021. Customer Lookalike Modeling: A Study of Machine Learning Techniques for Customer Lookalike Modeling. In Intelligent Data Communication Technologies and Internet of Things: Proceedings of ICICI 2020. Springer, 211-222."},{"key":"e_1_3_2_2_7_1","volume-title":"Proceedings of the 15th Conference of the European Chapter of the Association for Computational Linguistics","volume":"2","author":"Cossu Jean-Val\u00e8re","year":"2017","unstructured":"Jean-Val\u00e8re Cossu, Alejandro Molina-Villegas, and Mariana Tello-Signoret. 2017. Active Learning in Annotating Micro-Blogs Dealing with E-Reputation. In Proceedings of the 15th Conference of the European Chapter of the Association for Computational Linguistics: Volume 2, Short Papers. 12-18."},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.21236\/ADA440382"},{"key":"e_1_3_2_2_9_1","volume-title":"SimVLM: Simple Visual Language Model Pretraining with Weak Supervision. arXiv preprint arXiv:2108.10904","author":"Dai Zihang","year":"2021","unstructured":"Zihang Dai, Hanxiao Liu, Quoc V. Le, and Mingxing Tan. 2021. SimVLM: Simple Visual Language Model Pretraining with Weak Supervision. arXiv preprint arXiv:2108.10904 (2021)."},{"key":"e_1_3_2_2_10_1","first-page":"337","article-title":"Analysis of a greedy active learning strategy","author":"Dasgupta Sanjoy","year":"2005","unstructured":"Sanjoy Dasgupta. 2005. Analysis of a greedy active learning strategy. In Advances in Neural Information Processing Systems. 337-344.","journal-title":"Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-1423"},{"key":"e_1_3_2_2_12_1","unstructured":"Xin Dong Sen Jia and Hongyu Xiong. 2024. COEF-VQ: Cost-Efficient Video Quality Understanding through a Cascaded Multimodal LLM Framework. arXiv:2412.10435 [cs.CV] https:\/\/arxiv.org\/abs\/2412.10435"},{"key":"e_1_3_2_2_13_1","volume-title":"Proceedings of the 34th International Conference on Machine Learning. PMLR, 1183-1192","author":"Gal Yarin","year":"2017","unstructured":"Yarin Gal, Riashat Islam, and Zoubin Ghahramani. 2017. Deep Bayesian Active Learning with Image Data. In Proceedings of the 34th International Conference on Machine Learning. PMLR, 1183-1192."},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"crossref","unstructured":"Ruijie Hou Zhaoyang Yang Yu Ming Hongyu Lu Zhuobin Zheng Yu Chen Qinsong Zeng and Ming Chen. 2024. Cross Domain LifeLong Sequential Modeling for Online Click-Through Rate Prediction. arXiv:2312.06424 [cs.IR] https:\/\/arxiv.org\/abs\/2312.06424","DOI":"10.1145\/3637528.3671601"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.3115\/1614049.1614067"},{"key":"e_1_3_2_2_16_1","unstructured":"X. Li Y. Liu Y. Zhang and X. Wang. 2021. VLMo: Unified Vision-Language Pre-Training with Mixture-of-Modality-Experts. https:\/\/arxiv.org\/abs\/2111.02358"},{"key":"e_1_3_2_2_17_1","volume-title":"CAT: Enhancing Multimodal Large Language Model to Answer Questions in Dynamic Audio-Visual Scenarios. https: \/\/link.springer.com\/chapter\/10.1007\/978-3-031-72684-2_9","author":"Liu Jian","year":"2023","unstructured":"Jian Liu, Yu Lee, and Ming Zhang. 2023. CAT: Enhancing Multimodal Large Language Model to Answer Questions in Dynamic Audio-Visual Scenarios. https: \/\/link.springer.com\/chapter\/10.1007\/978-3-031-72684-2_9"},{"key":"e_1_3_2_2_18_1","unstructured":"Ming Liu Qi Zhang and Jian Li. 2023. Macaw-LLM: Multi-Modal Language Modeling with Image Audio Video and Text Integration. https:\/\/arxiv.org\/abs\/2306.09093"},{"key":"e_1_3_2_2_19_1","volume-title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach. arXiv preprint arXiv:1907.11692","author":"Liu Yinhan","year":"2019","unstructured":"Yinhan Liu, Myle Ott, Naman Goyal, Jingfei Du, Mandar Joshi, Danqi Chen, Omer Levy, Mike Lewis, Luke Zettlemoyer, and Veselin Stoyanov. 2019. RoBERTa: A Robustly Optimized BERT Pretraining Approach. arXiv preprint arXiv:1907.11692 (2019)."},{"key":"e_1_3_2_2_20_1","unstructured":"Yiming Liu Yi Zhang Furu Wei Wenliang Li Xipeng Zhou and Maosong Sun. 2022. Pretraining Methods for Sequence-to-Sequence Models. https:\/\/arxiv.org\/ abs\/2208.06366"},{"key":"e_1_3_2_2_21_1","unstructured":"Fredrik Olsson. 2009. A literature survey of active machine learning in the context of natural language processing. SICS Technical Report (2009)."},{"key":"e_1_3_2_2_22_1","volume-title":"Whisper: Robust Speech Recognition via Large-Scale Weak Supervision. https: \/\/cdn.openai.com\/papers\/whisper.pdf","author":"Radford Alec","year":"2022","unstructured":"Alec Radford, Jeff Wu, Rewon Child, David Luan, and Dario Amodei. 2022. Whisper: Robust Speech Recognition via Large-Scale Weak Supervision. https: \/\/cdn.openai.com\/papers\/whisper.pdf"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-44816-0_31"},{"key":"e_1_3_2_2_24_1","volume-title":"Garnett (Eds.)","volume":"31","author":"Sensoy Murat","year":"2018","unstructured":"Murat Sensoy, Lance Kaplan, and Melih Kandemir. 2018. Evidential Deep Learning to Quantify Classification Uncertainty. In Advances in Neural Information Processing Systems, S. Bengio, H. Wallach, H. Larochelle, K. Grauman, N. Cesa-Bianchi, and R. Garnett (Eds.), Vol. 31. Curran Associates, Inc. https:\/\/proceedings.neurips. cc\/paper_files\/paper\/2018\/file\/a981f2b708044d6fb4a71a1463242520-Paper.pdf"},{"key":"e_1_3_2_2_25_1","unstructured":"Burr Settles. 2010. Active learning literature survey. University of Wisconsin-Madison Department of Computer Sciences 52 (2010)."},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.3115\/1613715.1613855"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"crossref","unstructured":"Amit Sharma Hua Li Xue Li and Jian Jiao. 2024. Optimizing Novelty of Top-k Recommendations using Large Language Models and Reinforcement Learning. arXiv:2406.14169 [cs.IR] https:\/\/arxiv.org\/abs\/2406.14169","DOI":"10.1145\/3637528.3671618"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.3115\/1218955.1219030"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/2783258.2788603"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00607"},{"key":"e_1_3_2_2_31_1","volume-title":"Deep Interest with Hierarchical Attention Network for Click-Through Rate Prediction. arXiv preprint arXiv:2103.03436","author":"Sun Hao","year":"2021","unstructured":"Hao Sun, Guorui Zhou, Yingxia Shao, Dawei Feng, Siqi Ouyang, Qiwei Chen, Xiuqiang He, Jingren Zhou, Han Li, and Kun Gai. 2021. Deep Interest with Hierarchical Attention Network for Click-Through Rate Prediction. arXiv preprint arXiv:2103.03436 (2021)."},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2023.106742"},{"key":"e_1_3_2_2_33_1","first-page":"45","article-title":"Support vector machine active learning with applications to text classification","author":"Tong Simon","year":"2001","unstructured":"Simon Tong and Daphne Koller. 2001. Support vector machine active learning with applications to text classification. In Journal of Machine Learning Research. 45-66.","journal-title":"Journal of Machine Learning Research."},{"key":"e_1_3_2_2_34_1","volume-title":"Saksham Singhal, Subhojit Som, et al.","author":"Wang Wenhui","year":"2022","unstructured":"Wenhui Wang, Hangbo Bao, Li Dong, Johan Bjorck, Zhiliang Peng, Qiang Liu, Kriti Aggarwal, Owais Khan Mohammed, Saksham Singhal, Subhojit Som, et al. 2022. Image as a foreign language: Beit pretraining for all vision and visionlanguage tasks. arXiv preprint arXiv:2208.10442 (2022)."},{"key":"e_1_3_2_2_35_1","volume-title":"Multi-Grained Vision Language Pre-Training: Aligning Texts with Visual Concepts. arXiv preprint arXiv:2203.14962","author":"Ma Lin","year":"2022","unstructured":"WenguanWang, Lin Ma, Shaodi You, Nianyi Jiang, Jianqiang Zhang, Zhedong Hu, Jianbing Shen, and Ling Shao. 2022. Multi-Grained Vision Language Pre-Training: Aligning Texts with Visual Concepts. arXiv preprint arXiv:2203.14962 (2022)."},{"key":"e_1_3_2_2_36_1","volume-title":"Fast Bayesian force fields from active learning: study of inter-dimensional transformation of stanene. npj Computational Materials 7, 1","author":"Xie Yu","year":"2021","unstructured":"Yu Xie, Jonathan Vandermause, Lixin Sun, Andrea Cepellotti, and Boris Kozinsky. 2021. Fast Bayesian force fields from active learning: study of inter-dimensional transformation of stanene. npj Computational Materials 7, 1 (2021), 1-9."},{"key":"e_1_3_2_2_37_1","volume-title":"Conference on Learning Theory. 1343-1359","author":"Yan Yining","year":"2017","unstructured":"Yining Yan, Kamalika Chaudhuri, and Tara Javidi. 2017. Revisiting the sample complexity of active learning. In Conference on Learning Theory. 1343-1359."},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00018"},{"key":"e_1_3_2_2_39_1","volume-title":"VLP: Vision Language Planning for Autonomous Driving. In Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Yu Xin","year":"2020","unstructured":"Xin Yu, Xijun Wang, Mingkang Zhu, Linchao Bao, Wei Liu, and Dacheng Tao. 2020. VLP: Vision Language Planning for Autonomous Driving. In Conference on Computer Vision and Pattern Recognition (CVPR)."},{"key":"e_1_3_2_2_40_1","unstructured":"Li Zhao Yuxi Chen and Hao Wang. 2024. Towards Multimodal In-Context Learning for Vision & Language Models. https:\/\/arxiv.org\/abs\/2403.12736"},{"key":"e_1_3_2_2_41_1","volume-title":"Deep Interest Evolution Network for Click-Through Rate Prediction. arXiv preprint arXiv:1809.03672","author":"Zhou Guorui","year":"2019","unstructured":"Guorui Zhou, Xiaoqiang Zhu, Chenru Song, Ying Fan, Han Zhu, Xiao Ma, Yanghui Yan, Junqi Jin, Han Li, and Kun Gai. 2019. Deep Interest Evolution Network for Click-Through Rate Prediction. arXiv preprint arXiv:1809.03672 (2019)."},{"key":"e_1_3_2_2_42_1","volume-title":"MEERKAT: Audio-Visual Large Language Model for Grounding in Space and Time. https:\/\/link.springer.com\/chapter\/10.1007\/978-3-031-73039-9_4","author":"Zhu Qiang","year":"2024","unstructured":"Qiang Zhu, Qi Zhang, and Zhen Lee. 2024. MEERKAT: Audio-Visual Large Language Model for Grounding in Space and Time. https:\/\/link.springer.com\/chapter\/10.1007\/978-3-031-73039-9_4"}],"event":{"name":"KDD '25: The 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Toronto ON Canada","acronym":"KDD '25","sponsor":["SIGKDD ACM Special Interest Group on Knowledge Discovery in Data","SIGMOD ACM Special Interest Group on Management of Data"]},"container-title":["Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.2"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3711896.3737195","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T18:21:26Z","timestamp":1777573286000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3711896.3737195"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,3]]},"references-count":42,"alternative-id":["10.1145\/3711896.3737195","10.1145\/3711896"],"URL":"https:\/\/doi.org\/10.1145\/3711896.3737195","relation":{},"subject":[],"published":{"date-parts":[[2025,8,3]]},"assertion":[{"value":"2025-08-03","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}