{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:14:30Z","timestamp":1765340070054,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":59,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["No. 2022YFB36066"],"award-info":[{"award-number":["No. 2022YFB36066"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Shenzhen Science and Technology Project","award":["KJZD20240903103210014, JCYJ20220818101001004"],"award-info":[{"award-number":["KJZD20240903103210014, JCYJ20220818101001004"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3754560","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:38:54Z","timestamp":1761377934000},"page":"846-855","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Towards Fine-Grained Human Motion Video Captioning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-5803-5924","authenticated-orcid":false,"given":"Guorui","family":"Song","sequence":"first","affiliation":[{"name":"Tsinghua University, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-1815-2342","authenticated-orcid":false,"given":"Guocun","family":"Wang","sequence":"additional","affiliation":[{"name":"Tsinghua University, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-4722-8749","authenticated-orcid":false,"given":"Zhe","family":"Huang","sequence":"additional","affiliation":[{"name":"Tsinghua University, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-7549-6874","authenticated-orcid":false,"given":"Jing","family":"Lin","sequence":"additional","affiliation":[{"name":"ByteDance, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5005-7166","authenticated-orcid":false,"given":"Xuefei","family":"Zhe","sequence":"additional","affiliation":[{"name":"City University of Hong Kong, Hong Kong, Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-0466-5930","authenticated-orcid":false,"given":"Jian","family":"Li","sequence":"additional","affiliation":[{"name":"Tsinghua University, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2792-8469","authenticated-orcid":false,"given":"Haoqian","family":"Wang","sequence":"additional","affiliation":[{"name":"Tsinghua University, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Baldomero R. \u00c1rbol and Dan Casas. 2024. BodyShapeGPT: SMPL Body Shape Manipulation with LLMs. arXiv:2410.03556"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","unstructured":"Jinze Bai Shuai Bai Shusheng Yang Shijie Wang Sinan Tan Peng Wang Junyang Lin Chang Zhou and Jingren Zhou. 2023. Qwen-VL: A Versatile Vision-Language Model for Understanding Localization Text Reading and Beyond. arXiv:2308.12966 doi:10.48550\/arXiv.2308.12966","DOI":"10.48550\/arXiv.2308.12966"},{"key":"e_1_3_2_1_3_1","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Sibo Song Kai Dang Peng Wang Shijie Wang Jun Tang et al. 2025. Qwen2. 5-vl technical report. arXiv preprint arXiv:2502.13923 (2025)."},{"key":"e_1_3_2_1_4_1","volume-title":"Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization. 65-72","author":"Banerjee Satanjeev","year":"2005","unstructured":"Satanjeev Banerjee and Alon Lavie. 2005. METEOR: An automatic metric for MT evaluation with improved correlation with human judgments. In Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization. 65-72."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","unstructured":"Emanuele Bugliarello Anurag Arnab Roni Paiss Pieter-Jan Kindermans and Cordelia Schmid. 2025. What Are You Doing ? A Closer Look at Controllable Human Video Generation. arXiv:2503.04666 [cs] doi:10.48550\/arXiv.2503.04666","DOI":"10.48550\/arXiv.2503.04666"},{"key":"e_1_3_2_1_6_1","volume-title":"PerturboLLaVA: Reducing Multimodal Hallucinations with Perturbative Visual Training. In The Thirteenth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=j4LITBSUjs","author":"Chen Cong","year":"2025","unstructured":"Cong Chen, Mingyu Liu, Chenchen Jing, Yizhou Zhou, Fengyun Rao, Hao Chen, Bo Zhang, and Chunhua Shen. 2025. PerturboLLaVA: Reducing Multimodal Hallucinations with Perturbative Visual Training. In The Thirteenth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=j4LITBSUjs"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Lin Chen Jinsong Li Xiaoyi Dong Pan Zhang Yuhang Zang Zehui Chen Haodong Duan Jiaqi Wang Yu Qiao Dahua Lin et al. 2024a. Are We on the Right Way for Evaluating Large Vision-Language Models? arXiv preprint arXiv:2403.20330 (2024).","DOI":"10.52202\/079017-0850"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"crossref","unstructured":"Lin Chen Xilin Wei Jinsong Li Xiaoyi Dong Pan Zhang Yuhang Zang Zehui Chen Haodong Duan Bin Lin Zhenyu Tang et al. 2024c. ShareGPT4Video: Improving Video Understanding and Generation with Better Captions. arXiv preprint arXiv:2406.04325 (2024).","DOI":"10.52202\/079017-0614"},{"key":"e_1_3_2_1_9_1","volume-title":"MotionLLM: Understanding Human Behaviors from Human Motions and Videos. arXiv preprint arXiv:2405.20340","author":"Chen Ling-Hao","year":"2024","unstructured":"Ling-Hao Chen, Shunlin Lu, Ailing Zeng, Hao Zhang, Benyou Wang, Ruimao Zhang, and Lei Zhang. 2024b. MotionLLM: Understanding Human Behaviors from Human Motions and Videos. arXiv preprint arXiv:2405.20340 (2024)."},{"key":"e_1_3_2_1_10_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 24185-24198","author":"Chen Zhe","year":"2024","unstructured":"Zhe Chen, Jiannan Wu, Wenhai Wang, Weijie Su, Guo Chen, Sen Xing, Muyan Zhong, Qinglong Zhang, Xizhou Zhu, Lewei Lu, Bin Li, Ping Luo, Tong Lu, Yu Qiao, and Jifeng Dai. 2024d. InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 24185-24198."},{"key":"e_1_3_2_1_11_1","volume-title":"Mitigating hallucination in visual language models with visual supervision. arXiv preprint arXiv:2311.16479","author":"Chen Zhiyang","year":"2023","unstructured":"Zhiyang Chen, Yousong Zhu, Yufei Zhan, Zhaowen Li, Chaoyang Zhao, Jinqiao Wang, and Ming Tang. 2023. Mitigating hallucination in visual language models with visual supervision. arXiv preprint arXiv:2311.16479 (2023)."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","unstructured":"Yung-Sung Chuang Yujia Xie Hongyin Luo Yoon Kim James Glass and Pengcheng He. 2024. DoLa: Decoding by Contrasting Layers Improves Factuality in Large Language Models. arXiv:2309.03883 [cs] doi:10.48550\/arXiv.2309.03883","DOI":"10.48550\/arXiv.2309.03883"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"crossref","unstructured":"Delmas Ginger and Weinzaepfel Philippe and Lucas Thomas and Moreno-Noguer Francesc and Rogez Gr\u00e9gory. 2022. PoseScript: 3D Human Poses from Natural Language. In ECCV.","DOI":"10.1007\/978-3-031-20068-7_20"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"crossref","unstructured":"Delmas Ginger and Weinzaepfel Philippe and Moreno-Noguer Francesc and Rogez Gr\u00e9gory. 2023. PoseFix: Correcting 3D Human Poses with Natural Language. In ICCV.","DOI":"10.1109\/ICCV51070.2023.01379"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","unstructured":"Ailin Deng Tri Cao Zhirui Chen and Bryan Hooi. 2025. Words or Vision: Do Vision-Language Models Have Blind Faith in Text ? arXiv:2503.02199 [cs] doi:10.48550\/arXiv.2503.02199","DOI":"10.48550\/arXiv.2503.02199"},{"key":"e_1_3_2_1_16_1","unstructured":"Aaron Grattafiori Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Alex Vaughan et al. 2024. The llama 3 herd of models. arXiv preprint arXiv:2407.21783 (2024)."},{"key":"e_1_3_2_1_17_1","unstructured":"Kristen Grauman Andrew Westbury Lorenzo Torresani Kris Kitani Jitendra Malik Triantafyllos Afouras Kumar Ashutosh Vijay Baiyya"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00509"},{"key":"e_1_3_2_1_19_1","unstructured":"Daya Guo Dejian Yang Haowei Zhang Junxiao Song Ruoyu Zhang Runxin Xu Qihao Zhu Shirong Ma Peiyi Wang Xiao Bi et al. 2025. Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning. arXiv preprint arXiv:2501.12948 (2025)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","unstructured":"Lehan He Zeren Chen Zhelun Shi Tianyu Yu Jing Shao and Lu Sheng. 2024a. A Topic-level Self-Correctional Approach to Mitigate Hallucinations in MLLMs. arXiv:2411.17265 [cs] doi:10.48550\/arXiv.2411.17265","DOI":"10.48550\/arXiv.2411.17265"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"crossref","unstructured":"Xin He Longhui Wei Lingxi Xie and Qi Tian. 2024b. Incorporating Visual Experts to Resolve the Information Loss in Multimodal Large Language Models. arXiv:2401.03105 [cs.CV] https:\/\/arxiv.org\/abs\/2401.03105","DOI":"10.24963\/ijcai.2024\/123"},{"key":"e_1_3_2_1_22_1","volume-title":"Proceedings of the IEEE\/CVF international conference on computer vision. 6982-6991","author":"Hidalgo Gines","year":"2019","unstructured":"Gines Hidalgo, Yaadhav Raaj, Haroon Idrees, Donglai Xiang, Hanbyul Joo, Tomas Simon, and Yaser Sheikh. 2019. Single-network whole-body pose estimation. In Proceedings of the IEEE\/CVF international conference on computer vision. 6982-6991."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"crossref","unstructured":"Wenyi Hong* Yean Cheng* Zhuoyi Yang* Weihan Wang Lefan Wang Xiaotao Gu Shiyu Huang Yuxiao Dong and Jie Tang. 2024. MotionBench: Benchmarking and Improving Fine-grained Video Motion Understanding for Vision Language Models. arXiv:2501.02955 [cs.CV]","DOI":"10.1109\/CVPR52734.2025.00791"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2311.17911"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02553"},{"key":"e_1_3_2_1_26_1","unstructured":"Xuan Ju Yiming Gao Zhaoyang Zhang Ziyang Yuan Xintao Wang Ailing Zeng Yu Xiong Qiang Xu and Ying Shan. 2024. MiraData: A Large-Scale Video Dataset with Long Durations and Structured Captions. arXiv:2407.06358 [cs.CV] https:\/\/arxiv.org\/abs\/2407.06358"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01316"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","unstructured":"Bo Li Yuanhan Zhang Dong Guo Renrui Zhang Feng Li Hao Zhang Kaichen Zhang Yanwei Li Ziwei Liu and Chunyuan Li. 2024a. LLaVA-OneVision: Easy Visual Task Transfer. arXiv:2408.03326 doi:10.48550\/arXiv.2408.03326","DOI":"10.48550\/arXiv.2408.03326"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","unstructured":"Feng Li Renrui Zhang Hao Zhang Yuanhan Zhang Bo Li Wei Li Zejun Ma and Chunyuan Li. 2024b. LLaVA-NeXT-Interleave: Tackling Multi-Image Video and 3D in Large Multimodal Models. arXiv:2407.07895 [cs] doi:10.48550\/arXiv.2407.07895","DOI":"10.48550\/arXiv.2407.07895"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","unstructured":"Yanwei Li Chengyao Wang and Jiaya Jia. 2023. LLaMA-VID: An Image Is Worth 2 Tokens in Large Language Models. arXiv:2311.17043 [cs] doi:10.48550\/arXiv.2311.17043","DOI":"10.48550\/arXiv.2311.17043"},{"key":"e_1_3_2_1_31_1","volume-title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection. arXiv preprint arXiv:2311.10122","author":"Lin Bin","year":"2023","unstructured":"Bin Lin, Bin Zhu, Yang Ye, Munan Ning, Peng Jin, and Li Yuan. 2023b. Video-LLaVA: Learning United Visual Representation by Alignment Before Projection. arXiv preprint arXiv:2311.10122 (2023)."},{"key":"e_1_3_2_1_32_1","volume-title":"Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74-81.","author":"Lin Chin-Yew","year":"2004","unstructured":"Chin-Yew Lin. 2004. Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74-81."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2312.07533"},{"key":"e_1_3_2_1_34_1","volume-title":"Motion-X: A Large-scale 3D Expressive Whole-body Human Motion Dataset. Advances in Neural Information Processing Systems","author":"Lin Jing","year":"2023","unstructured":"Jing Lin, Ailing Zeng, Shunlin Lu, Yuanhao Cai, Ruimao Zhang, Haoqian Wang, and Lei Zhang. 2023a. Motion-X: A Large-scale 3D Expressive Whole-body Human Motion Dataset. Advances in Neural Information Processing Systems (2023)."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","unstructured":"Haotian Liu Chunyuan Li Yuheng Li and Yong Jae Lee. 2024. Improved Baselines with Visual Instruction Tuning. arXiv:2310.03744 [cs] doi:10.48550\/arXiv.2310.03744","DOI":"10.48550\/arXiv.2310.03744"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3596711.3596800"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","unstructured":"Muhammad Maaz Hanoona Rasheed Salman Khan and Fahad Shahbaz Khan. 2024. Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models. arXiv:2306.05424 [cs] doi:10.48550\/arXiv.2306.05424","DOI":"10.48550\/arXiv.2306.05424"},{"volume-title":"International Conference on Computer Vision. 5442-5451","author":"Mahmood Naureen","key":"e_1_3_2_1_38_1","unstructured":"Naureen Mahmood, Nima Ghorbani, Nikolaus F. Troje, Gerard Pons-Moll, and Michael J. Black. 2019. AMASS: Archive of Motion Capture as Surface Shapes. In International Conference on Computer Vision. 5442-5451."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","unstructured":"Haozhou Pang Tianwei Ding Lanshan He and Qi Gan. 2025. Global Position Aware Group Choreography Using Large Language Model. arXiv:2503.09645 [cs] doi:10.48550\/arXiv.2503.09645","DOI":"10.48550\/arXiv.2503.09645"},{"key":"e_1_3_2_1_40_1","volume-title":"Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311-318","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. Bleu: a method for automatic evaluation of machine translation. In Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311-318."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1089\/big.2016.0028"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2103.00020"},{"key":"e_1_3_2_1_43_1","unstructured":"Istv\u00e1n S\u00e1r\u00e1ndi and Gerard Pons-Moll. 2024. Neural Localizer Fields for Continuous 3D Human Pose and Shape Estimation. (2024)."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00914"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","unstructured":"Chao Wang Xuancheng Zhou Weiwei Fu and Yang Zhou. 2025b. Mitigating Hallucinations in Large Vision-Language Models with Internal Fact-Based Contrastive Decoding. arXiv:2502.01056 [cs] doi:10.48550\/arXiv.2502.01056","DOI":"10.48550\/arXiv.2502.01056"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","unstructured":"Peng Wang Shuai Bai Sinan Tan Shijie Wang Zhihao Fan Jinze Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Yang Fan Kai Dang Mengfei Du Xuancheng Ren Rui Men Dayiheng Liu Chang Zhou Jingren Zhou and Junyang Lin. 2024. Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution. arXiv:2409.12191 [cs] doi:10.48550\/arXiv.2409.12191","DOI":"10.48550\/arXiv.2409.12191"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","unstructured":"Yin Wang Mu Li Jiapeng Liu Zhiying Leng Frederick W. B. Li Ziyao Zhang and Xiaohui Liang. 2025a. Fg-T2M: LLMs-Augmented Fine-Grained Text Driven Human Motion Generation. arXiv:2502.05534 [cs] doi:10.48550\/arXiv.2502.05534","DOI":"10.48550\/arXiv.2502.05534"},{"key":"e_1_3_2_1_49_1","volume-title":"Vitpose: Simple vision transformer baselines for human pose estimation. Advances in neural information processing systems","author":"Xu Yufei","year":"2022","unstructured":"Yufei Xu, Jing Zhang, Qiming Zhang, and Dacheng Tao. 2022. Vitpose: Simple vision transformer baselines for human pose estimation. Advances in neural information processing systems, Vol. 35 (2022), 38571-38584."},{"key":"e_1_3_2_1_50_1","unstructured":"An Yang Baosong Yang Binyuan Hui Bo Zheng Bowen Yu Chang Zhou Chengpeng Li Chengyuan Li Dayiheng Liu Fei Huang Guanting Dong Haoran Wei Huan Lin Jialong Tang Jialin Wang Jian Yang Jianhong Tu Jianwei Zhang Jianxin Ma Jianxin Yang Jin Xu Jingren Zhou Jinze Bai Jinzheng He Junyang Lin Kai Dang Keming Lu Keqin Chen Kexin Yang Mei Li Mingfeng Xue Na Ni Pei Zhang Peng Wang Ru Peng Rui Men Ruize Gao Runji Lin Shijie Wang Shuai Bai Sinan Tan Tianhang Zhu Tianhao Li Tianyu Liu Wenbin Ge Xiaodong Deng Xiaohuan Zhou Xingzhang Ren Xinyu Zhang Xipin Wei Xuancheng Ren Xuejing Liu Yang Fan Yang Yao Yichang Zhang Yu Wan Yunfei Chu Yuqiong Liu Zeyu Cui Zhenru Zhang Zhifang Guo and Zhihao Fan. 2024. Qwen2 Technical Report. arXiv:2407.10671 [cs.CL] https:\/\/arxiv.org\/abs\/2407.10671"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW60793.2023.00455"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2310.16045"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01100"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","unstructured":"Hang Zhang Xin Li and Lidong Bing. 2023. Video-LLaMA: An Instruction-Tuned Audio-Visual Language Model for Video Understanding. arXiv:2306.02858 [cs] doi:10.48550\/arXiv.2306.02858","DOI":"10.48550\/arXiv.2306.02858"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","unstructured":"Jinrui Zhang Teng Wang Haigang Zhang Ping Lu and Feng Zheng. 2024a. Reflective Instruction Tuning: Mitigating Hallucinations in Large Vision-Language Models. arXiv:2407.11422 [cs] doi:10.48550\/arXiv.2407.11422","DOI":"10.48550\/arXiv.2407.11422"},{"key":"e_1_3_2_1_56_1","volume-title":"Motion-X: A Large-Scale Multimodal 3D Whole-body Human Motion Dataset. arXiv preprint arXiv:2501.05098","author":"Zhang Yuhong","year":"2025","unstructured":"Yuhong Zhang, Jing Lin, Ailing Zeng, Guanlin Wu, Shunlin Lu, Yurong Fu, Yuanhao Cai, Ruimao Zhang, Haoqian Wang, and Lei Zhang. 2025. Motion-X: A Large-Scale Multimodal 3D Whole-body Human Motion Dataset. arXiv preprint arXiv:2501.05098 (2025)."},{"key":"e_1_3_2_1_57_1","unstructured":"Yuanhan Zhang Jinming Wu Wei Li Bo Li Zejun Ma Ziwei Liu and Chunyuan Li. 2024b. Video Instruction Tuning With Synthetic Data. arXiv:2410.02713 [cs.CV] https:\/\/arxiv.org\/abs\/2410.02713"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","unstructured":"Zijia Zhao Yuqi Huo Tongtian Yue Longteng Guo Haoyu Lu Bingning Wang Weipeng Chen and Jing Liu. 2025. Efficient Motion-Aware Video MLLM. arXiv:2503.13016 [cs] doi:10.48550\/arXiv.2503.13016","DOI":"10.48550\/arXiv.2503.13016"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","unstructured":"Xin Zou Yizhou Wang Yibo Yan Sirui Huang Kening Zheng Junkai Chen Chang Tang and Xuming Hu. 2024. Look Twice Before You Answer: Memory-Space Visual Retracing for Hallucination Mitigation in Multimodal Large Language Models. arXiv:2410.03577 [cs] doi:10.48550\/arXiv.2410.03577","DOI":"10.48550\/arXiv.2410.03577"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3754560","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:10:03Z","timestamp":1765339803000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3754560"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":59,"alternative-id":["10.1145\/3746027.3754560","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3754560","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}