{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T05:05:24Z","timestamp":1765343124949,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":74,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3754752","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:27:39Z","timestamp":1761377259000},"page":"10787-10796","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["NEXUS-O: An Omni-Perceptive and -Interactive Model for Language, Audio, and Vision"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-3738-7998","authenticated-orcid":false,"given":"Che","family":"Liu","sequence":"first","affiliation":[{"name":"Imperial College London, London, United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1499-3309","authenticated-orcid":false,"given":"Yingji","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of Manchester, Manchester, United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-8811-8454","authenticated-orcid":false,"given":"Dong","family":"Zhang","sequence":"additional","affiliation":[{"name":"HiThink Research, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-3352-652X","authenticated-orcid":false,"given":"Weijie","family":"Zhang","sequence":"additional","affiliation":[{"name":"HiThink Research, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-6755-5405","authenticated-orcid":false,"given":"Chenggong","family":"Gong","sequence":"additional","affiliation":[{"name":"HiThink Research, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-9695-6241","authenticated-orcid":false,"given":"Yu","family":"Lu","sequence":"additional","affiliation":[{"name":"HiThink Research, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-4500-3963","authenticated-orcid":false,"given":"Shilin","family":"Zhou","sequence":"additional","affiliation":[{"name":"Soochow University, Suzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-7450-9887","authenticated-orcid":false,"given":"Ziliang","family":"Gan","sequence":"additional","affiliation":[{"name":"HiThink Research, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8019-2334","authenticated-orcid":false,"given":"Ziao","family":"Wang","sequence":"additional","affiliation":[{"name":"Hong Kong Baptist University, Hong Kong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-0386-8479","authenticated-orcid":false,"given":"Haipang","family":"Wu","sequence":"additional","affiliation":[{"name":"HiThink Research, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9421-4100","authenticated-orcid":false,"given":"Ji","family":"Liu","sequence":"additional","affiliation":[{"name":"HiThink Research, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4430-4837","authenticated-orcid":false,"given":"Andre","family":"Freitas","sequence":"additional","affiliation":[{"name":"University of Manchester, Manchester, United Kingdom and Idiap Research Institute, Martigny, Switzerland"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7570-5756","authenticated-orcid":false,"given":"Qifan","family":"Wang","sequence":"additional","affiliation":[{"name":"Meta AI, Menlo Park, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5550-6461","authenticated-orcid":false,"given":"Zenglin","family":"Xu","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1823-2726","authenticated-orcid":false,"given":"Rongjunchen","family":"Zhang","sequence":"additional","affiliation":[{"name":"HiThink Research, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3041-5851","authenticated-orcid":false,"given":"Yong","family":"Dai","sequence":"additional","affiliation":[{"name":"X-Humanoid, Beijing, China and Fudan University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Nguyen Bach, Amit Bahree, Arash Bakhtiari, Jianmin Bao, Harkirat Behl, et al.","author":"Abdin Marah","year":"2024","unstructured":"Marah Abdin, Jyoti Aneja, Hany Awadalla, Ahmed Awadallah, Ammar Ahmad Awan, Nguyen Bach, Amit Bahree, Arash Bakhtiari, Jianmin Bao, Harkirat Behl, et al. 2024. Phi-3 technical report: A highly capable language model locally on your phone. arXiv preprint arXiv:2404.14219 (2024)."},{"key":"e_1_3_2_1_2_1","unstructured":"Philip Anastassiou Jiawei Chen Jitong Chen Yuanzhe Chen Zhuo Chen Ziyi Chen Jian Cong Lelai Deng Chuang Ding Lu Gao Mingqing Gong Peisong Huang Qingqing Huang Zhiying Huang Yuanyuan Huo Dongya Jia Chumin Li Feiya Li Hui Li Jiaxin Li Xiaoyang Li Xingxing Li Lin Liu Shouda Liu Sichao Liu Xudong Liu Yuchen Liu Zhengxi Liu Lu Lu Junjie Pan Xin Wang Yuping Wang Yuxuan Wang Zhen Wei Jian Wu Chao Yao Yifeng Yang Yuanhao Yi Junteng Zhang Qidi Zhang Shuo Zhang Wenjie Zhang Yang Zhang Zilin Zhao Dejian Zhong and Xiaobin Zhuang. 2024. Seed-TTS: A Family of HighQuality Versatile Speech Generation Models. arXiv:2406.02430 [eess.AS] https:\/\/arxiv.org\/abs\/2406.02430"},{"key":"e_1_3_2_1_3_1","volume-title":"Common voice: A massively-multilingual speech corpus. arXiv preprint arXiv:1912.06670","author":"Ardila Rosana","year":"2019","unstructured":"Rosana Ardila, Megan Branson, Kelly Davis, Michael Henretty, Michael Kohler, Josh Meyer, Reuben Morais, Lindsay Saunders, Francis M Tyers, and Gregor Weber. 2019. Common voice: A massively-multilingual speech corpus. arXiv preprint arXiv:1912.06670 (2019)."},{"key":"e_1_3_2_1_4_1","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Sibo Song Kai Dang Peng Wang Shijie Wang Jun Tang Humen Zhong Yuanzhi Zhu Mingkun Yang Zhaohai Li Jianqiang Wan Pengfei Wang Wei Ding Zheren Fu Yiheng Xu Jiabo Ye Xi Zhang Tianbao Xie Zesen Cheng Hang Zhang Zhibo Yang Haiyang Xu and Junyang Lin. 2025. Qwen2.5-VL Technical Report. arXiv:2502.13923 [cs.CV] https:\/\/arxiv.org\/abs\/2502.13923"},{"key":"e_1_3_2_1_5_1","first-page":"LL","volume":"201","author":"Bu Hui","unstructured":"Hui Bu, Jiayu Du, Xingyu Na, Bengu Wu, and Hao Zheng. 2017. AISHELL-1: An Open-Source Mandarin Speech Corpus and A Speech Recognition Baseline. arXiv:1709.05522 [cs.CL] https:\/\/arxiv.org\/abs\/1709.05522","journal-title":"Hao Zheng."},{"key":"e_1_3_2_1_6_1","volume-title":"EMOVA: Empowering Language Models to See, Hear and Speak with Vivid Emotions. arXiv:2409.18042 [cs.CV] https:\/\/arxiv.org\/abs\/2409.18042","author":"Chen Kai","year":"2025","unstructured":"Kai Chen, Yunhao Gou, Runhui Huang, Zhili Liu, Daxin Tan, Jing Xu, Chunwei Wang, Yi Zhu, Yihan Zeng, Kuo Yang, Dingdong Wang, Kun Xiang, Haoyuan Li, Haoli Bai, Jianhua Han, Xiaohui Li, Weike Jin, Nian Xie, Yu Zhang, James T. Kwok, Hengshuang Zhao, Xiaodan Liang, Dit-Yan Yeung, Xiao Chen, Zhenguo Li, Wei Zhang, Qun Liu, Jun Yao, Lanqing Hong, Lu Hou, and Hang Xu. 2025. EMOVA: Empowering Language Models to See, Hear and Speak with Vivid Emotions. arXiv:2409.18042 [cs.CV] https:\/\/arxiv.org\/abs\/2409.18042"},{"key":"e_1_3_2_1_7_1","unstructured":"Qian Chen Yafeng Chen Yanni Chen Mengzhe Chen Yingda Chen Chong Deng Zhihao Du Ruize Gao Changfeng Gao Zhifu Gao et al. 2025. MinMo: A Multimodal Large Language Model for Seamless Voice Interaction. arXiv preprint arXiv:2501.06282 (2025)."},{"key":"e_1_3_2_1_8_1","unstructured":"Qian Chen Yafeng Chen Yanni Chen Mengzhe Chen Yingda Chen Chong Deng Zhihao Du Ruize Gao Changfeng Gao Zhifu Gao Yabin Li Xiang Lv Jiaqing Liu Haoneng Luo Bin Ma Chongjia Ni Xian Shi Jialong Tang Hui Wang Hao Wang Wen Wang Yuxuan Wang Yunlan Xu Fan Yu Zhijie Yan Yexin Yang Baosong Yang Xian Yang Guanrou Yang Tianyu Zhao Qinglin Zhang Shiliang Zhang Nan Zhao Pei Zhang Chong Zhang and Jinren Zhou. 2025. MinMo: A Multimodal Large Language Model for Seamless Voice Interaction. arXiv:2501.06282 [cs.CL] https:\/\/arxiv.org\/abs\/2501.06282"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"crossref","unstructured":"Zhe Chen Weiyun Wang Hao Tian Shenglong Ye Zhangwei Gao Erfei Cui Wenwen Tong Kongzhi Hu Jiapeng Luo Zheng Ma Ji Ma Jiaqi Wang Xiaoyi Dong Hang Yan Hewei Guo Conghui He Botian Shi Zhenjiang Jin Chao Xu Bin Wang Xingjian Wei Wei Li Wenjian Zhang Bo Zhang Pinlong Cai Licheng Wen Xiangchao Yan Min Dou Lewei Lu Xizhou Zhu Tong Lu Dahua Lin Yu Qiao Jifeng Dai and Wenhai Wang. 2024. How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites. arXiv:2404.16821 [cs.CV] https:\/\/arxiv.org\/abs\/2404.16821","DOI":"10.1007\/s11432-024-4231-5"},{"key":"e_1_3_2_1_10_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 24185--24198","author":"Chen Zhe","year":"2024","unstructured":"Zhe Chen, Jiannan Wu, Wenhai Wang, Weijie Su, Guo Chen, Sen Xing, Muyan Zhong, Qinglong Zhang, Xizhou Zhu, Lewei Lu, et al. 2024. Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 24185--24198."},{"key":"e_1_3_2_1_11_1","unstructured":"Yunfei Chu Jin Xu Qian Yang Haojie Wei Xipin Wei Zhifang Guo Yichong Leng Yuanjun Lv Jinzheng He Junyang Lin et al. 2024. Qwen2-audio technical report. arXiv preprint arXiv:2407.10759 (2024)."},{"key":"e_1_3_2_1_12_1","unstructured":"Yunfei Chu Jin Xu Qian Yang Haojie Wei Xipin Wei Zhifang Guo Yichong Leng Yuanjun Lv Jinzheng He Junyang Lin Chang Zhou and Jingren Zhou. 2024. Qwen2-Audio Technical Report. arXiv:2407.10759 [eess.AS] https:\/\/arxiv.org\/abs\/2407.10759"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/SLT54892.2023.10023141"},{"key":"e_1_3_2_1_14_1","volume-title":"Simple and controllable music generation. Advances in Neural Information Processing Systems 36","author":"Copet Jade","year":"2023","unstructured":"Jade Copet, Felix Kreuk, Itai Gat, Tal Remez, David Kant, Gabriel Synnaeve, Yossi Adi, and Alexandre D\u00e9fossez. 2023. Simple and controllable music generation. Advances in Neural Information Processing Systems 36 (2023)."},{"key":"e_1_3_2_1_15_1","volume-title":"One model, multiple modalities: A sparsely activated approach for text, sound, image, video and code. arXiv preprint arXiv:2205.06126","author":"Dai Yong","year":"2022","unstructured":"Yong Dai, Duyu Tang, Liangxin Liu, Minghuan Tan, Cong Zhou, Jingquan Wang, Zhangyin Feng, Fan Zhang, Xueyu Hu, and Shuming Shi. 2022. One model, multiple modalities: A sparsely activated approach for text, sound, image, video and code. arXiv preprint arXiv:2205.06126 (2022)."},{"key":"e_1_3_2_1_16_1","volume-title":"Moshi: a speech-text foundation model for real-time dialogue. arXiv preprint arXiv:2410.00037","author":"D\u00e9fossez Alexandre","year":"2024","unstructured":"Alexandre D\u00e9fossez, Laurent Mazar\u00e9, Manu Orsini, Am\u00e9lie Royer, Patrick P\u00e9rez, Herv\u00e9 J\u00e9gou, Edouard Grave, and Neil Zeghidour. 2024. Moshi: a speech-text foundation model for real-time dialogue. arXiv preprint arXiv:2410.00037 (2024)."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_18_1","unstructured":"Xiaoyi Dong Pan Zhang Yuhang Zang Yuhang Cao Bin Wang Linke Ouyang Xilin Wei Songyang Zhang Haodong Duan Maosong Cao Wenwei Zhang Yining Li Hang Yan Yang Gao Xinyue Zhang Wei Li Jingwen Li Kai Chen Conghui He Xingcheng Zhang Yu Qiao Dahua Lin and Jiaqi Wang. 2024. InternLM-XComposer2: Mastering Free-form Text-Image Composition and Comprehension in Vision-Language Large Model. arXiv:2401.16420 [cs.CV] https:\/\/arxiv.org\/abs\/2401.16420"},{"key":"e_1_3_2_1_19_1","first-page":"ll","volume":"201","author":"Du Jiayu","unstructured":"Jiayu Du, Xingyu Na, Xuechen Liu, and Hui Bu. 2018. Aishell-2: Transforming mandarin asr research into industrial scale. arXiv preprint arXiv:1808.10583 (2018).","journal-title":"Hui Bu."},{"key":"e_1_3_2_1_20_1","first-page":"LL","volume":"201","author":"Du Jiayu","unstructured":"Jiayu Du, Xingyu Na, Xuechen Liu, and Hui Bu. 2018. AISHELL-2: Transforming Mandarin ASR Research Into Industrial Scale. arXiv:1808.10583 [cs.CL] https:\/\/arxiv.org\/abs\/1808.10583","journal-title":"Hui Bu."},{"key":"e_1_3_2_1_21_1","volume-title":"Cosyvoice: A scalable multilingual zero-shot text-to-speech synthesizer based on supervised semantic tokens. arXiv preprint arXiv:2407.05407","author":"Du Zhihao","year":"2024","unstructured":"Zhihao Du, Qian Chen, Shiliang Zhang, Kai Hu, Heng Lu, Yexin Yang, Hangrui Hu, Siqi Zheng, Yue Gu, Ziyang Ma, et al. 2024. Cosyvoice: A scalable multilingual zero-shot text-to-speech synthesizer based on supervised semantic tokens. arXiv preprint arXiv:2407.05407 (2024)."},{"key":"e_1_3_2_1_22_1","unstructured":"Zhihao Du Qian Chen Shiliang Zhang Kai Hu Heng Lu Yexin Yang Hangrui Hu Siqi Zheng Yue Gu Ziyang Ma Zhifu Gao and Zhijie Yan. 2024. CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens. arXiv:2407.05407 [cs.SD] https:\/\/arxiv.org\/abs\/2407.05407"},{"key":"e_1_3_2_1_23_1","unstructured":"Zhihao Du Yuxuan Wang Qian Chen Xian Shi Xiang Lv Tianyu Zhao Zhifu Gao Yexin Yang Changfeng Gao Hui Wang Fan Yu Huadai Liu Zhengyan Sheng Yue Gu Chong Deng Wen Wang Shiliang Zhang Zhijie Yan and Jingren Zhou. 2024. CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models. arXiv:2412.10117 [cs.SD] https:\/\/arxiv.org\/abs\/2412.10117"},{"key":"e_1_3_2_1_24_1","volume-title":"Llama-omni: Seamless speech interaction with large language models. arXiv preprint arXiv:2409.06666","author":"Fang Qingkai","year":"2024","unstructured":"Qingkai Fang, Shoutao Guo, Yan Zhou, Zhengrui Ma, Shaolei Zhang, and Yang Feng. 2024. Llama-omni: Seamless speech interaction with large language models. arXiv preprint arXiv:2409.06666 (2024)."},{"key":"e_1_3_2_1_25_1","unstructured":"Qingkai Fang Shoutao Guo Yan Zhou Zhengrui Ma Shaolei Zhang and Yang Feng. 2024. LLaMA-Omni: Seamless Speech Interaction with Large Language Models. arXiv:2409.06666 [cs.CL] https:\/\/arxiv.org\/abs\/2409.06666"},{"key":"e_1_3_2_1_26_1","volume-title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models. arXiv:2306.13394 [cs.CV] https:\/\/arxiv.org\/abs\/2306.13394","author":"Fu Chaoyou","year":"2024","unstructured":"Chaoyou Fu, Peixian Chen, Yunhang Shen, Yulei Qin, Mengdan Zhang, Xu Lin, Jinrui Yang, Xiawu Zheng, Ke Li, Xing Sun, Yunsheng Wu, and Rongrong Ji. 2024. MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models. arXiv:2306.13394 [cs.CV] https:\/\/arxiv.org\/abs\/2306.13394"},{"key":"e_1_3_2_1_27_1","unstructured":"Chaoyou Fu Yuhan Dai Yongdong Luo Lei Li Shuhuai Ren Renrui Zhang Zihan Wang Chenyu Zhou Yunhang Shen Mengdan Zhang Peixian Chen Yanwei Li Shaohui Lin Sirui Zhao Ke Li Tong Xu Xiawu Zheng Enhong Chen Rongrong Ji and Xing Sun. 2024. Video-MME: The First-Ever Comprehensive Evaluation Benchmark of Multi-modal LLMs in Video Analysis. arXiv:2405.21075 [cs.CV] https:\/\/arxiv.org\/abs\/2405.21075"},{"key":"e_1_3_2_1_28_1","volume-title":"Vita: Towards opensource interactive omni multimodal llm. arXiv preprint arXiv:2408.05211","author":"Fu Chaoyou","year":"2024","unstructured":"Chaoyou Fu, Haojia Lin, Zuwei Long, Yunhang Shen, Meng Zhao, Yifan Zhang, Shaoqi Dong, Xiong Wang, Di Yin, Long Ma, et al. 2024. Vita: Towards opensource interactive omni multimodal llm. arXiv preprint arXiv:2408.05211 (2024)."},{"key":"e_1_3_2_1_29_1","unstructured":"Chaoyou Fu Haojia Lin Xiong Wang Yi-Fan Zhang Yunhang Shen Xiaoyu Liu Haoyu Cao Zuwei Long Heting Gao Ke Li Long Ma Xiawu Zheng Rongrong Ji Xing Sun Caifeng Shan and Ran He. 2025. VITA-1.5: Towards GPT-4o Level Real-Time Vision and Speech Interaction. arXiv:2501.01957 [cs.CV] https:\/\/arxiv.org\/abs\/2501.01957"},{"key":"e_1_3_2_1_30_1","unstructured":"Chaoyou Fu Haojia Lin Xiong Wang Yi-Fan Zhang Yunhang Shen Xiaoyu Liu Yangze Li Zuwei Long Heting Gao Ke Li et al. 2025. VITA-1.5: Towards GPT-4o Level Real-Time Vision and Speech Interaction. arXiv preprint arXiv:2501.01957 (2025)."},{"key":"e_1_3_2_1_31_1","unstructured":"Ling Fu Biao Yang Zhebin Kuang Jiajun Song Yuzhe Li Linghao Zhu Qidi Luo Xinyu Wang Hao Lu Mingxin Huang Zhang Li Guozhi Tang Bin Shan Chunhui Lin Qi Liu Binghong Wu Hao Feng Hao Liu Can Huang Jingqun Tang Wei Chen Lianwen Jin Yuliang Liu and Xiang Bai. 2024. OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning. arXiv:2501.00321 [cs.CV] https:\/\/arxiv.org\/abs\/2501.00321"},{"key":"e_1_3_2_1_32_1","volume-title":"Keith Achorn, Anjali Gopi, David Kanter, Maximilian Lam, Mark Mazumder, and Vijay Janapa Reddi.","author":"Galvez Daniel","year":"2021","unstructured":"Daniel Galvez, Greg Diamos, Juan Ciro, Juan Felipe Cer\u00f3n, Keith Achorn, Anjali Gopi, David Kanter, Maximilian Lam, Mark Mazumder, and Vijay Janapa Reddi. 2021. The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage. arXiv:2111.09344 [cs.LG] https:\/\/arxiv.org\/abs\/2111.09344"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"crossref","unstructured":"Tianrui Guan Fuxiao Liu Xiyang Wu Ruiqi Xian Zongxia Li Xiaoyu Liu Xijun Wang Lichang Chen Furong Huang Yaser Yacoob et al. 2023. HallusionBench: An Advanced Diagnostic Suite for Entangled Language Hallucination and Visual Illusion in Large Vision-Language Models. arXiv preprint arXiv:2310.14566 (2023).","DOI":"10.1109\/CVPR52733.2024.01363"},{"key":"e_1_3_2_1_34_1","volume-title":"Proceedings of the 41st International Conference on Machine Learning (Proceedings of Machine Learning Research","volume":"20642","author":"Huh Minyoung","year":"2024","unstructured":"Minyoung Huh, Brian Cheung, Tongzhou Wang, and Phillip Isola. 2024. Position: The Platonic Representation Hypothesis. In Proceedings of the 41st International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 235), Ruslan Salakhutdinov, Zico Kolter, Katherine Heller, Adrian Weller, Nuria Oliver, Jonathan Scarlett, and Felix Berkenkamp (Eds.). PMLR, 20617--20642. https:\/\/proceedings.mlr.press\/v235\/huh24a.html"},{"key":"e_1_3_2_1_35_1","unstructured":"Jared Kaplan Sam McCandlish Tom Henighan Tom B. Brown Benjamin Chess Rewon Child Scott Gray Alec Radford Jeffrey Wu and Dario Amodei. 2020. Scaling Laws for Neural Language Models. arXiv:2001.08361 [cs.LG] https:\/\/arxiv.org\/abs\/2001.08361"},{"key":"e_1_3_2_1_36_1","volume-title":"International conference on machine learning. PMLR, 3519--3529","author":"Kornblith Simon","year":"2019","unstructured":"Simon Kornblith, Mohammad Norouzi, Honglak Lee, and Geoffrey Hinton. 2019. Similarity of neural network representations revisited. In International conference on machine learning. PMLR, 3519--3529."},{"key":"e_1_3_2_1_37_1","unstructured":"Boxun Li Yadong Li Zhiyuan Li Congyi Liu Weilin Liu Guowei Niu Zheyue Tan Haiyang Xu Zhuyu Yao Tao Yuan Dong Zhou Yueqing Zhuang Shengen Yan Guohao Dai and Yu Wang. 2025. Megrez-Omni Technical Report. arXiv:2502.15803 [cs.LG] https:\/\/arxiv.org\/abs\/2502.15803"},{"key":"e_1_3_2_1_38_1","unstructured":"Yadong Li Haoze Sun Mingan Lin Tianpeng Li Guosheng Dong Tao Zhang Bowen Ding Wei Song Zhenglin Cheng Yuqi Huo et al. 2024. Baichuan-omni technical report. arXiv preprint arXiv:2410.08565 2 3 (2024)."},{"key":"e_1_3_2_1_39_1","unstructured":"Yizhi Li Ge Zhang Yinghao Ma Ruibin Yuan Kang Zhu Hangyu Guo Yiming Liang Jiaheng Liu Zekun Wang Jian Yang Siwei Wu Xingwei Qu Jinjie Shi Xinyue Zhang Zhenzhu Yang Xiangzhou Wang Zhaoxiang Zhang Zachary Liu Emmanouil Benetos Wenhao Huang and Chenghua Lin. 2024. OmniBench: Towards The Future of Universal Omni-Language Models. arXiv:2409.15272 [cs.CL] https:\/\/arxiv.org\/abs\/2409.15272"},{"key":"e_1_3_2_1_40_1","unstructured":"Shijia Liao Yuxuan Wang Tianyu Li Yifan Cheng Ruoyi Zhang Rongzhi Zhou and Yijin Xing. 2024. Fish-Speech: Leveraging Large Language Models for Advanced Multilingual Text-to-Speech Synthesis. arXiv:2411.01156 [cs.SD] https:\/\/arxiv.org\/abs\/2411.01156"},{"key":"e_1_3_2_1_41_1","volume-title":"VILA: On Pre-training for Visual Language Models. arXiv:2312.07533 [cs.CV] https:\/\/arxiv.org\/abs\/2312. 07533","author":"Lin Ji","year":"2024","unstructured":"Ji Lin, Hongxu Yin, Wei Ping, Yao Lu, Pavlo Molchanov, Andrew Tao, Huizi Mao, Jan Kautz, Mohammad Shoeybi, and Song Han. 2024. VILA: On Pre-training for Visual Language Models. arXiv:2312.07533 [cs.CV] https:\/\/arxiv.org\/abs\/2312. 07533"},{"key":"e_1_3_2_1_42_1","unstructured":"Haotian Liu Chunyuan Li Yuheng Li Bo Li Yuanhan Zhang Sheng Shen and Yong Jae Lee. 2024. LLaVA-NeXT: Improved reasoning OCR and world knowledge. https:\/\/llava-vl.github.io\/blog\/2024-01--30-llava-next\/"},{"key":"e_1_3_2_1_43_1","unstructured":"Haotian Liu Chunyuan Li Qingyang Wu and Yong Jae Lee. 2023. Visual Instruction Tuning."},{"key":"e_1_3_2_1_44_1","volume-title":"Mathvista: Evaluating mathematical reasoning of foundation models in visual contexts. arXiv preprint arXiv:2310.02255","author":"Lu Pan","year":"2023","unstructured":"Pan Lu, Hritik Bansal, Tony Xia, Jiacheng Liu, Chunyuan Li, Hannaneh Hajishirzi, Hao Cheng, Kai-Wei Chang, Michel Galley, and Jianfeng Gao. 2023. Mathvista: Evaluating mathematical reasoning of foundation models in visual contexts. arXiv preprint arXiv:2310.02255 (2023)."},{"key":"e_1_3_2_1_45_1","volume-title":"Ovis: Structural Embedding Alignment for Multimodal Large Language Model. arXiv:2405.20797 [cs.CV] https:\/\/arxiv.org\/abs\/2405.20797","author":"Lu Shiyin","year":"2024","unstructured":"Shiyin Lu, Yang Li, Qing-Guo Chen, Zhao Xu, Weihua Luo, Kaifu Zhang, and Han-Jia Ye. 2024. Ovis: Structural Embedding Alignment for Multimodal Large Language Model. arXiv:2405.20797 [cs.CV] https:\/\/arxiv.org\/abs\/2405.20797"},{"key":"e_1_3_2_1_46_1","unstructured":"Run Luo Ting-En Lin Haonan Zhang Yuchuan Wu Xiong Liu Min Yang Yongbin Li Longze Chen Jiaming Li Lei Zhang Yangyi Chen Hamid Alinejad-Rokny and Fei Huang. 2025. OpenOmni: Large Language Models Pivot Zero-shot Omnimodal Alignment across Language with Real-time Self-Aware Emotional Speech Synthesis. arXiv:2501.04561 [cs.CL] https:\/\/arxiv.org\/abs\/2501.04561"},{"key":"e_1_3_2_1_47_1","unstructured":"Meta. 2024. Llama3.2 technical report. (2024)."},{"key":"e_1_3_2_1_48_1","volume-title":"Spoken Question Answering and Speech Continuation Using Spectrogram-Powered LLM. arXiv preprint arXiv:2305.15255","author":"Nachmani Eliya","year":"2023","unstructured":"Eliya Nachmani, Alon Levkovitch, Roy Hirsch, Julian Salazar, Chulayuth Asawaroengchai, Soroosh Mariooryad, Ehud Rivlin, RJ Skerry-Ryan, and Michelle Tadmor Ramanovich. 2023. Spoken Question Answering and Speech Continuation Using Spectrogram-Powered LLM. arXiv preprint arXiv:2305.15255 (2023)."},{"key":"e_1_3_2_1_49_1","unstructured":"Eliya Nachmani Alon Levkovitch Roy Hirsch Julian Salazar Chulayuth Asawaroengchai Soroosh Mariooryad Ehud Rivlin RJ Skerry-Ryan and Michelle Tadmor Ramanovich. 2024. Spoken Question Answering and Speech Continuation Using Spectrogram-Powered LLM. arXiv:2305.15255 [cs.CL] https:\/\/arxiv.org\/abs\/2305.15255"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"crossref","unstructured":"Patrick K. O'Neill Vitaly Lavrukhin Somshubra Majumdar Vahid Noroozi Yuekai Zhang Oleksii Kuchaiev Jagadeesh Balam Yuliya Dovzhenko Keenan Freyberg Michael D. Shulman Boris Ginsburg Shinji Watanabe and Georg Kucsko. 2021. SPGISpeech: 5 000 hours of transcribed financial audio for fully formatted end-to-end speech recognition. arXiv:2104.02014 [cs.CL] https:\/\/arxiv.org\/abs\/2104.02014","DOI":"10.21437\/Interspeech.2021-1860"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"e_1_3_2_1_53_1","volume-title":"Mls: A large-scale multilingual dataset for speech research. arXiv preprint arXiv:2012.03411","author":"Pratap Vineel","year":"2020","unstructured":"Vineel Pratap, Qiantong Xu, Anuroop Sriram, Gabriel Synnaeve, and Ronan Collobert. 2020. Mls: A large-scale multilingual dataset for speech research. arXiv preprint arXiv:2012.03411 (2020)."},{"key":"e_1_3_2_1_54_1","volume-title":"MLS: A Large-Scale Multilingual Dataset for Speech Research. ArXiv abs\/2012.03411","author":"Pratap Vineel","year":"2020","unstructured":"Vineel Pratap, Qiantong Xu, Anuroop Sriram, Gabriel Synnaeve, and Ronan Collobert. 2020. MLS: A Large-Scale Multilingual Dataset for Speech Research. ArXiv abs\/2012.03411 (2020)."},{"key":"e_1_3_2_1_55_1","volume-title":"Tao Xu, Greg Brockman, Christine McLeavey, and Ilya Sutskever.","author":"Radford Alec","year":"2022","unstructured":"Alec Radford, Jong Wook Kim, Tao Xu, Greg Brockman, Christine McLeavey, and Ilya Sutskever. 2022. Robust Speech Recognition via Large-Scale Weak Supervision. arXiv:2212.04356 [eess.AS] https:\/\/arxiv.org\/abs\/2212.04356"},{"key":"e_1_3_2_1_56_1","volume-title":"International conference on machine learning. PMLR, 28492--28518","author":"Radford Alec","year":"2023","unstructured":"Alec Radford, Jong Wook Kim, Tao Xu, Greg Brockman, Christine McLeavey, and Ilya Sutskever. 2023. Robust speech recognition via large-scale weak supervision. In International conference on machine learning. PMLR, 28492--28518."},{"key":"e_1_3_2_1_57_1","volume-title":"Svcca: Singular vector canonical correlation analysis for deep learning dynamics and interpretability. Advances in neural information processing systems 30","author":"Raghu Maithra","year":"2017","unstructured":"Maithra Raghu, Justin Gilmer, Jason Yosinski, and Jascha Sohl-Dickstein. 2017. Svcca: Singular vector canonical correlation analysis for deep learning dynamics and interpretability. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3463257"},{"key":"e_1_3_2_1_59_1","volume-title":"Jihan Yang, Shusheng Yang, Adithya Iyer, Xichen Pan, Ziteng Wang, Rob Fergus, Yann LeCun, and Saining Xie.","author":"Tong Shengbang","year":"2024","unstructured":"Shengbang Tong, Ellis Brown, Penghao Wu, Sanghyun Woo, Manoj Middepogu, Sai Charitha Akula, Jihan Yang, Shusheng Yang, Adithya Iyer, Xichen Pan, Ziteng Wang, Rob Fergus, Yann LeCun, and Saining Xie. 2024. Cambrian-1: A Fully Open, Vision-Centric Exploration of Multimodal LLMs. arXiv:2406.16860 [cs.CV] https:\/\/arxiv.org\/abs\/2406.16860"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"crossref","unstructured":"Changhan Wang Anne Wu and Juan Pino. 2020. CoVoST 2 and Massively Multilingual Speech-to-Text Translation. arXiv:2007.10310 [cs.CL] https:\/\/arxiv.org\/abs\/2007.10310","DOI":"10.21437\/Interspeech.2021-2027"},{"key":"e_1_3_2_1_61_1","unstructured":"Peng Wang Shuai Bai Sinan Tan Shijie Wang Zhihao Fan Jinze Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge et al. 2024. Qwen2-vl: Enhancing vision-language model's perception of the world at any resolution. arXiv preprint arXiv:2409.12191 (2024)."},{"key":"e_1_3_2_1_62_1","volume-title":"Mini-omni2: Towards open-source gpt4o with vision, speech and duplex capabilities. arXiv preprint arXiv:2410.11190","author":"Xie Zhifei","year":"2024","unstructured":"Zhifei Xie and Changqiao Wu. 2024. Mini-omni2: Towards open-source gpt4o with vision, speech and duplex capabilities. arXiv preprint arXiv:2410.11190 (2024)."},{"key":"e_1_3_2_1_63_1","volume-title":"Mini-omni2: Towards open-source gpt4o with vision, speech and duplex capabilities. arXiv preprint arXiv:2410.11190","author":"Xie Zhifei","year":"2024","unstructured":"Zhifei Xie and Changqiao Wu. 2024. Mini-omni2: Towards open-source gpt4o with vision, speech and duplex capabilities. arXiv preprint arXiv:2410.11190 (2024)."},{"key":"e_1_3_2_1_64_1","unstructured":"Jin Xu Zhifang Guo Jinzheng He Hangrui Hu Ting He Shuai Bai Keqin Chen Jialin Wang Yang Fan Kai Dang Bin Zhang Xiong Wang Yunfei Chu and Junyang Lin. 2025. Qwen2.5-Omni Technical Report. arXiv:2503.20215 [cs.CL] https:\/\/arxiv.org\/abs\/2503.20215"},{"key":"e_1_3_2_1_65_1","unstructured":"An Yang Baosong Yang Beichen Zhang Binyuan Hui Bo Zheng Bowen Yu Chengyuan Li Dayiheng Liu Fei Huang Haoran Wei et al. 2024. Qwen2. 5 Technical Report. arXiv preprint arXiv:2412.15115 (2024)."},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"crossref","unstructured":"Qian Yang Jin Xu Wenrui Liu Yunfei Chu Ziyue Jiang Xiaohuan Zhou Yichong Leng Yuanjun Lv Zhou Zhao Chang Zhou et al. 2024. AIR-Bench: Benchmarking Large Audio-Language Models via Generative Comprehension. arXiv preprint arXiv:2402.07729 (2024).","DOI":"10.18653\/v1\/2024.acl-long.109"},{"key":"e_1_3_2_1_67_1","unstructured":"Yuan Yao Tianyu Yu Ao Zhang Chongyi Wang Junbo Cui Hongji Zhu Tianchi Cai Haoyu Li Weilin Zhao Zhihui He et al. 2024. MiniCPM-V: A GPT-4V Level MLLM on Your Phone. arXiv preprint arXiv:2408.01800 (2024)."},{"key":"e_1_3_2_1_68_1","unstructured":"Rong Ye Chengqi Zhao Tom Ko Chutong Meng Tao Wang Mingxuan Wang and Jun Cao. 2023. GigaST: A 10 000-hour Pseudo Speech Translation Corpus. arXiv:2204.03939 [cs.CL] https:\/\/arxiv.org\/abs\/2204.03939"},{"key":"e_1_3_2_1_69_1","unstructured":"Weihao Yu Zhengyuan Yang Linjie Li Jianfeng Wang Kevin Lin Zicheng Liu Xinchao Wang and Lijuan Wang. 2024. MM-Vet: Evaluating Large Multimodal Models for Integrated Capabilities. arXiv:2308.02490 [cs.AI] https:\/\/arxiv.org\/abs\/2308.02490"},{"key":"e_1_3_2_1_70_1","volume-title":"MMMU: A Massive Multidiscipline Multimodal Understanding and Reasoning Benchmark for Expert AGI. arXiv:2311.16502 [cs.CL] https:\/\/arxiv.org\/abs\/2311.16502","author":"Yue Xiang","year":"2024","unstructured":"Xiang Yue, Yuansheng Ni, Kai Zhang, Tianyu Zheng, Ruoqi Liu, Ge Zhang, Samuel Stevens, Dongfu Jiang, Weiming Ren, Yuxuan Sun, Cong Wei, Botao Yu, Ruibin Yuan, Renliang Sun, Ming Yin, Boyuan Zheng, Zhenzhu Yang, Yibo Liu, Wenhao Huang, Huan Sun, Yu Su, and Wenhu Chen. 2024. MMMU: A Massive Multidiscipline Multimodal Understanding and Reasoning Benchmark for Expert AGI. arXiv:2311.16502 [cs.CL] https:\/\/arxiv.org\/abs\/2311.16502"},{"key":"e_1_3_2_1_71_1","unstructured":"Aohan Zeng Zhengxiao Du Mingdao Liu Lei Zhang Shengmin Jiang Yuxiao Dong and Jie Tang. 2024. Scaling Speech-Text Pre-training with Synthetic Interleaved Data. arXiv:2411.17607 [cs.CL] https:\/\/arxiv.org\/abs\/2411.17607"},{"key":"e_1_3_2_1_72_1","volume-title":"Wenetspeech: A 10000 hours multi-domain mandarin corpus for speech recognition. In ICASSP.","author":"Zhang Binbin","year":"2022","unstructured":"Binbin Zhang, Hang Lv, Pengcheng Guo, Qijie Shao, Chao Yang, Lei Xie, Xin Xu, Hui Bu, Xiaoyu Chen, Chenchen Zeng, et al. 2022. Wenetspeech: A 10000 hours multi-domain mandarin corpus for speech recognition. In ICASSP."},{"key":"e_1_3_2_1_73_1","volume-title":"SpeechGPT: Empowering Large Language Models with Intrinsic Cross-Modal Conversational Abilities. arXiv preprint arXiv:2305.11000","author":"Zhang Dong","year":"2023","unstructured":"Dong Zhang, Shimin Li, Xin Zhang, Jun Zhan, Pengyu Wang, Yaqian Zhou, and Xipeng Qiu. 2023. SpeechGPT: Empowering Large Language Models with Intrinsic Cross-Modal Conversational Abilities. arXiv preprint arXiv:2305.11000 (2023)."},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-naacl.32"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3754752","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T05:01:40Z","timestamp":1765342900000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3754752"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":74,"alternative-id":["10.1145\/3746027.3754752","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3754752","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}