{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:15:06Z","timestamp":1765340106084,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":45,"publisher":"ACM","funder":[{"name":"National Natural Science Foundation of China"},{"DOI":"10.13039\/501100001809","name":"Guangdong Ba- sic and Applied Basic Research Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"CCF-Tencent Rhino-Bird Open Research Fund"},{"name":"Guangdong Provincial Project"},{"name":"Guangzhou Municipal Science and Technology Project"},{"name":"Guangdong Provincial Key Laboratory of Integrated Communications, Sensing and Computation for Ubiquitous Internet of Things"},{"name":"Guangzhou Municipal Key Laboratory on Future Networked Systems"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755423","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:47:18Z","timestamp":1761374838000},"page":"8313-8321","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Cued-Agent: A Collaborative Multi-Agent System for Automatic Cued Speech Recognition"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1977-3647","authenticated-orcid":false,"given":"Guanjie","family":"Huang","sequence":"first","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0135-7098","authenticated-orcid":false,"given":"Danny H.K.","family":"Tsang","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4464-146X","authenticated-orcid":false,"given":"Shan","family":"Yang","sequence":"additional","affiliation":[{"name":"Tencent AI Lab, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-6136-7619","authenticated-orcid":false,"given":"Guangzhi","family":"Lei","sequence":"additional","affiliation":[{"name":"Tencent AI Lab, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4497-0135","authenticated-orcid":false,"given":"Li","family":"Liu","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology (Guangzhou), Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","first-page":"3","article-title":"Cued Speech","volume":"112","author":"Cornett R. Orin","year":"1967","unstructured":"R. Orin Cornett. 1967. Cued Speech. American Annals of the Deaf, Vol. 112, 1 (1967), 3-13.","journal-title":"American Annals of the Deaf"},{"key":"e_1_3_2_1_2_1","first-page":"19","article-title":"Adapting Cued Speech to additional languages","volume":"5","author":"Cornett R. Orin","year":"1994","unstructured":"R. Orin Cornett. 1994. Adapting Cued Speech to additional languages. Cued Speech Journal, Vol. 5 (1994), 19-29.","journal-title":"Cued Speech Journal"},{"key":"e_1_3_2_1_3_1","volume-title":"Videoagent: A memory-augmented multimodal agent for video understanding. In ECCV.","author":"Fan Yue","year":"2024","unstructured":"Yue Fan, Xiaojian Ma, Rujie Wu, Yuntao Du, Jiaqi Li, Zhi Gao, and Qing Li. 2024. Videoagent: A memory-augmented multimodal agent for video understanding. In ECCV."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","unstructured":"Lufei Gao Shan Huang and Li Liu. 2023. A novel interpretable and generalizable re-synchronization model for Cued Speech based on a multi-cuer corpus. In Interspeech.","DOI":"10.21437\/Interspeech.2023-663"},{"volume-title":"Supervised sequence labelling with recurrent neural networks","author":"Graves Alex","key":"e_1_3_2_1_5_1","unstructured":"Alex Graves. 2012. Supervised sequence labelling. In Supervised sequence labelling with recurrent neural networks. Springer, 5-13."},{"key":"e_1_3_2_1_6_1","unstructured":"Daya Guo Dejian Yang Haowei Zhang Junxiao Song Ruoyu Zhang Runxin Xu Qihao Zhu Shirong Ma Peiyi Wang Xiao Bi et al. 2025. DeepSeek-R1: Incentivizing reasoning capability in LLMs via reinforcement learning. arXiv (2025)."},{"key":"e_1_3_2_1_7_1","volume-title":"Large language model based multi-agents: A survey of progress and challenges. arXiv","author":"Guo Taicheng","year":"2024","unstructured":"Taicheng Guo, Xiuying Chen, Yaqi Wang, Ruidi Chang, Shichao Pei, Nitesh V Chawla, Olaf Wiest, and Xiangliang Zhang. 2024. Large language model based multi-agents: A survey of progress and challenges. arXiv (2024)."},{"key":"e_1_3_2_1_8_1","volume-title":"LLM multi-agent systems: Challenges and open problems. arXiv","author":"Han Shanshan","year":"2024","unstructured":"Shanshan Han, Qifan Zhang, Yuhang Yao, Weizhao Jin, Zhaozhuo Xu, and Chaoyang He. 2024. LLM multi-agent systems: Challenges and open problems. arXiv (2024)."},{"key":"e_1_3_2_1_9_1","volume-title":"LLM-Based multi-agent systems for software engineering: literature review, vision and the road ahead. ACM Transactions on Software Engineering and Methodology","author":"He Junda","year":"2024","unstructured":"Junda He, Christoph Treude, and David Lo. 2024. LLM-Based multi-agent systems for software engineering: literature review, vision and the road ahead. ACM Transactions on Software Engineering and Methodology (2024)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2009.2016011"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2010.03.001"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2010.03.001"},{"key":"e_1_3_2_1_13_1","unstructured":"Panikos Heracleous Denis Beautemps and Norihiro Hagita. 2012. Continuous phoneme recognition in Cued Speech for French. In EUSIPCO."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1080\/02533839.1999.9670495"},{"key":"e_1_3_2_1_15_1","volume-title":"Danny Hin Kwok Tsang, and Li Liu","author":"Huang Guanjie","year":"2025","unstructured":"Guanjie Huang, Danny Hin Kwok Tsang, and Li Liu. 2025. Lend a hand: Semi training-free Cued Speech recognition via MLLM-driven hand modeling for barrier-free communication. arXiv (2025)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"crossref","unstructured":"Rongjie Huang Mingze Li Dongchao Yang Jiatong Shi Xuankai Chang Zhenhui Ye Yuning Wu Zhiqing Hong Jiawei Huang Jinglin Liu et al. 2024. AudioGPT: Understanding and generating speech music sound and talking head. In AAAI.","DOI":"10.1609\/aaai.v38i21.30570"},{"key":"e_1_3_2_1_17_1","volume-title":"MCCD: Multi-agent collaboration-based compositional diffusion for complex text-to-image generation. In CVPR.","author":"Li Mingcheng","year":"2025","unstructured":"Mingcheng Li, Xiaolu Hou, Ziyang Liu, Dingkang Yang, Ziyun Qian, Jiawei Chen, Jinjie Wei, Yue Jiang, Qingyao Xu, and Lihua Zhang. 2025. MCCD: Multi-agent collaboration-based compositional diffusion for complex text-to-image generation. In CVPR."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1007\/s44336-024-00009-2"},{"key":"e_1_3_2_1_19_1","volume-title":"Anim-director: A large multimodal model powered agent for controllable animation video generation. In SIGGRAPH Asia.","author":"Li Yunxin","year":"2024","unstructured":"Yunxin Li, Haoyuan Shi, Baotian Hu, Longyue Wang, Jiashun Zhu, Jinyi Xu, Zhen Zhao, and Min Zhang. 2024a. Anim-director: A large multimodal model powered agent for controllable animation video generation. In SIGGRAPH Asia."},{"key":"e_1_3_2_1_20_1","volume-title":"ICLR Workshop.","author":"Liang Jinhua","year":"2024","unstructured":"Jinhua Liang, Huan Zhang, Haohe Liu, Yin Cao, Qiuqiang Kong, Xubo Liu, Wenwu Wang, Mark D Plumbley, Huy Phan, and Emmanouil Benetos. 2024. WavCraft: Audio editing and generation with natural language prompts. In ICLR Workshop."},{"key":"e_1_3_2_1_21_1","volume-title":"Cued speech: An evaluative study. American Annals of the deaf","author":"Ling Daniel","year":"1975","unstructured":"Daniel Ling and Bryan R Clarke. 1975. Cued speech: An evaluative study. American Annals of the deaf (1975), 480-488."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1353\/aad.2019.0031"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2020.2976493"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"crossref","unstructured":"Li Liu Thomas Hueber Gang Feng and Denis Beautemps. 2018. Visual recognition of continuous Cued Speech using a tandem CNN-HMM approach. In Interspeech.","DOI":"10.21437\/Interspeech.2018-2434"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"crossref","unstructured":"Lei Liu and Li Liu. 2023. Cross-modal mutual learning for Cued Speech recognition. In ICASSP.","DOI":"10.1109\/ICASSP49357.2023.10095271"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2024.3363446"},{"key":"e_1_3_2_1_27_1","volume-title":"Juhyun Lee, Wan-Teh Chang, Wei Hua, Manfred Georg, and Matthias Grundmann.","author":"Lugaresi Camillo","year":"2019","unstructured":"Camillo Lugaresi, Jiuqiang Tang, Hadon Nash, Chris McClanahan, Esha Uboweja, Michael Hays, Fan Zhang, Chuo-Ling Chang, Ming Guang Yong, Juhyun Lee, Wan-Teh Chang, Wei Hua, Manfred Georg, and Matthias Grundmann. 2019. MediaPipe: A framework for building perception pipelines. arXiv (2019)."},{"key":"e_1_3_2_1_28_1","volume-title":"Auto-avsr: Audio-visual speech recognition with automatic labels. In ICASSP.","author":"Ma Pingchuan","year":"2023","unstructured":"Pingchuan Ma, Alexandros Haliassos, Adriana Fernandez-Lopez, Honglie Chen, Stavros Petridis, and Maja Pantic. 2023. Auto-avsr: Audio-visual speech recognition with automatic labels. In ICASSP."},{"key":"e_1_3_2_1_29_1","unstructured":"OpenAI. 2024. Hello GPT-4o. https:\/\/openai.com\/index\/hello-gpt-4o\/"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"crossref","unstructured":"Katerina Papadimitriou and Gerasimos Potamianos. 2021. A fully convolutional sequence learning approach for Cued Speech recognition from videos. In EUSIPCO.","DOI":"10.23919\/Eusipco47968.2020.9287365"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"crossref","unstructured":"Nils Reimers and Iryna Gurevych. 2019. Sentence-BERT: Sentence embeddings using siamese BERT-networks. In EMNLP.","DOI":"10.18653\/v1\/D19-1410"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"crossref","unstructured":"Sanjana Sankar Denis Beautemps and Thomas Hueber. 2022. Multistream neural architectures for Cued Speech recognition using a pre-trained visual feature extractor and constrained CTC decoding. In ICASSP.","DOI":"10.1109\/ICASSP43922.2022.9746976"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"crossref","unstructured":"Sefik Ilkin Serengil and Alper Ozpinar. 2020. LightFace: A hybrid deep face recognition framework. In ASYU.","DOI":"10.1109\/ASYU50717.2020.9259802"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3352811"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1093\/deafed\/enx026"},{"key":"e_1_3_2_1_36_1","volume-title":"SPAgent: adaptive task decomposition and model selection for general video generation and editing. arXiv","author":"Tu Rong-Cheng","year":"2024","unstructured":"Rong-Cheng Tu, Wenhao Sun, Zhao Jin, Jingyi Liao, Jiaxing Huang, and Dacheng Tao. 2024. SPAgent: adaptive task decomposition and model selection for general video generation and editing. arXiv (2024)."},{"key":"e_1_3_2_1_37_1","volume-title":"Multi-agent systems. Foundations of artificial intelligence","author":"der Hoek Wiebe Van","year":"2008","unstructured":"Wiebe Van der Hoek and Michael Wooldridge. 2008. Multi-agent systems. Foundations of artificial intelligence, Vol. 3 (2008), 887-928."},{"key":"e_1_3_2_1_38_1","volume-title":"Advances in NeurIPS","volume":"30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in NeurIPS, Vol. 30 (2017)."},{"key":"e_1_3_2_1_39_1","volume-title":"An attention self-supervised contrastive learning based three-stage model for hand shape feature representation in Cued Speech. arXiv","author":"Wang Jianrong","year":"2021","unstructured":"Jianrong Wang, Nan Gu, Mei Yu, Xuewei Li, Qiang Fang, and Li Liu. 2021a. An attention self-supervised contrastive learning based three-stage model for hand shape feature representation in Cued Speech. arXiv (2021)."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"crossref","unstructured":"Jianrong Wang Ge Zhang Zhenyu Wu Xuewei Li and Li Liu. 2021b. Self-supervised depth estimation via implicit cues from videos. In ICASSP.","DOI":"10.1109\/ICASSP39728.2021.9413407"},{"key":"e_1_3_2_1_41_1","first-page":"128374","article-title":"Genartist: Multimodal llm as an agent for unified image generation and editing","volume":"37","author":"Wang Zhenyu","year":"2024","unstructured":"Zhenyu Wang, Aoxue Li, Zhenguo Li, and Xihui Liu. 2024. Genartist: Multimodal llm as an agent for unified image generation and editing. Advances in NeurIPS, Vol. 37 (2024), 128374-128395.","journal-title":"Advances in NeurIPS"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2017.2763455"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2024.3370039"},{"key":"e_1_3_2_1_44_1","volume-title":"Long-video audio synthesis with multi-agent collaboration. arXiv","author":"Zhang Yehang","year":"2025","unstructured":"Yehang Zhang, Xinli Xu, Xiaojie Xu, Li Liu, and Yingcong Chen. 2025. Long-video audio synthesis with multi-agent collaboration. arXiv (2025)."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"crossref","unstructured":"Tianchen Zhou Zhongjie Duan Cen Chen Wenmeng Zhou Yanhao Wang and Yaliang Li. 2025. AgentStory: A multi-agent system for story visualization with multi-subject consistent text-to-image generation. In ICMR.","DOI":"10.1145\/3731715.3733271"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755423","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:11:35Z","timestamp":1765339895000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755423"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":45,"alternative-id":["10.1145\/3746027.3755423","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755423","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}