{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T15:55:02Z","timestamp":1781538902541,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":69,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T00:00:00Z","timestamp":1781481600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"Shenzhen Medical Research Fund","award":["A2503006"],"award-info":[{"award-number":["A2503006"]}]},{"name":"Shenzhen Polytechnic University Research Fund","award":["6025310023K"],"award-info":[{"award-number":["6025310023K"]}]},{"name":"National Natural Science Foundation of China","award":["62501412"],"award-info":[{"award-number":["62501412"]}]},{"name":"Science and Technology Development Fund, Macau SAR","award":["0193\/2023\/RIA3"],"award-info":[{"award-number":["0193\/2023\/RIA3"]}]},{"name":"Science and Technology Development Fund, Macau SAR","award":["0079\/2025\/AFJ"],"award-info":[{"award-number":["0079\/2025\/AFJ"]}]},{"name":"University of Macau","award":["MYRG-GRG2024-00065-FST-UMDF"],"award-info":[{"award-number":["MYRG-GRG2024-00065-FST-UMDF"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,16]]},"DOI":"10.1145\/3805622.3810570","type":"proceedings-article","created":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T14:42:57Z","timestamp":1781534577000},"page":"2561-2570","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["ATRIE: Adaptive Tuning for Robust Inference and Emotion in Persona-Driven Speech Synthesis"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-2385-5343","authenticated-orcid":false,"given":"Aoduo","family":"Li","sequence":"first","affiliation":[{"name":"Guangdong University of Technology, Jieyang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-3310-6835","authenticated-orcid":false,"given":"Haoran","family":"Lv","sequence":"additional","affiliation":[{"name":"Guangdong University of Technology, Jieyang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-4463-3364","authenticated-orcid":false,"given":"Hongjian","family":"Xu","sequence":"additional","affiliation":[{"name":"Guangdong University of Technology, Jieyang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-6511-7652","authenticated-orcid":false,"given":"Shengmin","family":"Li","sequence":"additional","affiliation":[{"name":"South China University of Technology, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-0097-5297","authenticated-orcid":false,"given":"Sihao","family":"Qin","sequence":"additional","affiliation":[{"name":"South China University of Technology, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2798-3134","authenticated-orcid":false,"given":"Zimeng","family":"Li","sequence":"additional","affiliation":[{"name":"Shenzhen Polytechnic University, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1788-3746","authenticated-orcid":false,"given":"Chi Man","family":"Pun","sequence":"additional","affiliation":[{"name":"University of Macau, Macau, Macao"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6000-3914","authenticated-orcid":false,"given":"Xuhang","family":"Chen","sequence":"additional","affiliation":[{"name":"Huizhou University, Huizhou, China and University of Macau, Macau, Macao"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,15]]},"reference":[{"key":"e_1_3_3_1_2_2","volume-title":"NeurIPS","author":"Baevski Alexei","year":"2020","unstructured":"Alexei Baevski, Yuhao Zhou, Abdelrahman Mohamed, and Michael Auli. 2020. wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Representations. In NeurIPS."},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"crossref","unstructured":"Zal\u00e1n Borsos Rapha\u00ebl Marinier Damien Vincent Eugene Kharitonov Olivier Pietquin Matthew Sharifi Dominik Roblek Olivier Teboul David Grangier Marco Tagliasacchi and Neil Zeghidour. 2023. AudioLM: A Language Modeling Approach to Audio Generation. IEEE\/ACM Transactions on Audio Speech and Language Processing 31 (2023) 2523\u20132533.","DOI":"10.1109\/TASLP.2023.3288409"},{"key":"e_1_3_3_1_4_2","unstructured":"Shuochen Chang Xiaofeng Zhang Qingyang Liu and Li Niu. 2025. D3ToM: Decider-Guided Dynamic Token Merging for Accelerating Diffusion MLLMs. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2511.12280 (2025)."},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"crossref","unstructured":"Sanyuan Chen Chengyi Wang Zhengyang Chen Yu Wu Shujie Liu Zhuo Chen Jinyu Li Naoyuki Kanda Takuya Yoshioka Xiong Xiao Jian Wu Long Zhou Shuo Ren Yanmin Qian Yao Qian Jian Wu Michael Zeng Xiangzhan Yu and Furu Wei. 2022. WavLM: Large-Scale Self-Supervised Pre-Training for Full Stack Speech Processing. IEEE Journal of Selected Topics in Signal Processing 16 6 (2022) 1505\u20131518.","DOI":"10.1109\/JSTSP.2022.3188113"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.313"},{"key":"e_1_3_3_1_7_2","first-page":"1086","volume-title":"Interspeech","author":"Chung Joon\u00a0Son","year":"2018","unstructured":"Joon\u00a0Son Chung, Arsha Nagrani, and Andrew Zisserman. 2018. VoxCeleb2: Deep Speaker Recognition. In Interspeech. 1086\u20131090."},{"key":"e_1_3_3_1_8_2","unstructured":"Alibaba Cloud. 2023. Qwen-7B: A Towering Language Model. https:\/\/github.com\/QwenLM\/Qwen."},{"key":"e_1_3_3_1_9_2","first-page":"3830","volume-title":"Interspeech","author":"Desplanques Brecht","year":"2020","unstructured":"Brecht Desplanques, Jenthe Thienpondt, and Kris Demuynck. 2020. ECAPA-TDNN: Emphasized Channel Attention, Propagation and Aggregation in TDNN Based Speaker Verification. In Interspeech. 3830\u20133834."},{"key":"e_1_3_3_1_10_2","unstructured":"Zhihao Du Qian Chen Shiliang Zhang Kai Hu Heng Lu Yexin Yang Hangrui Hu Siqi Zheng Yue Gu Ziyang Ma Zhifu Gao and Zhijie Yan. 2024. CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.05407 (2024)."},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095889"},{"key":"e_1_3_3_1_12_2","unstructured":"Peng Gao Yujian Lee Xiaofeng Zhang Zailong Chen and Hui Zhang. 2025. Remember Me: Bridging the Long-Range Gap in LVLMs with Three-Step Inference-Only Decay Resilience Strategies. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2511.09868 (2025)."},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"crossref","unstructured":"Wei-Ning Hsu Benjamin Bolte Yao-Hung\u00a0Hubert Tsai Kushal Lakhotia Ruslan Salakhutdinov and Abdelrahman Mohamed. 2021. HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units. IEEE\/ACM Transactions on Audio Speech and Language Processing 29 (2021) 3451\u20133460.","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"e_1_3_3_1_14_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10447880"},{"key":"e_1_3_3_1_15_2","first-page":"3605","volume-title":"Interspeech","author":"Jeong Myeonghun","year":"2021","unstructured":"Myeonghun Jeong, Hyeongju Kim, Sung\u00a0Jun Cheon, Byoung\u00a0Jin Choi, and Nam\u00a0Soo Kim. 2021. Diff-TTS: A Denoising Diffusion Model for Text-to-Speech. In Interspeech. 3605\u20133609."},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"crossref","unstructured":"Hao Jiang Sen Li Weihuang Liu Hongjin Zheng Jinghao Liu and Yang Zhang. 2020. Geometry-aware cell detection with deep learning. Msystems 5 1 (2020) 10\u20131128.","DOI":"10.1128\/msystems.00840-19"},{"key":"e_1_3_3_1_17_2","unstructured":"Ziyue Jiang Jinglin Liu Yi Ren Jinzheng He Chen Zhang Zhenhui Ye Pengfei Wei Chunfeng Wang Xiang Yin Zejun Ma and Zhou Zhao. 2023. Mega-TTS 2: Zero-Shot Text-to-Speech with Arbitrary Length Speech Prompts. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2307.07218 (2023)."},{"key":"e_1_3_3_1_18_2","first-page":"22605","volume-title":"ICML","author":"Ju Zeqian","year":"2024","unstructured":"Zeqian Ju, Yuancheng Wang, Kai Shen, Xu Tan, Detai Xin, Dongchao Yang, Eric Liu, Yichong Leng, Kaitao Song, Siliang Tang, Zhizheng Wu, Tao Qin, Xiangyang Li, Wei Ye, Shikun Zhang, Jiang Bian, Lei He, Jinyu Li, and Sheng Zhao. 2024. NaturalSpeech 3: Zero-Shot Speech Synthesis with Factorized Codec and Diffusion Models. In ICML. 22605\u201322623."},{"key":"e_1_3_3_1_19_2","first-page":"5530","volume-title":"ICML","author":"Kim Jaehyeon","year":"2021","unstructured":"Jaehyeon Kim, Jungil Kong, and Juhee Son. 2021. Conditional Variational Autoencoder with Adversarial Learning for End-to-End Text-to-Speech. In ICML. 5530\u20135540."},{"key":"e_1_3_3_1_20_2","volume-title":"NeurIPS","author":"Kong Jungil","year":"2020","unstructured":"Jungil Kong, Jaehyeon Kim, and Jaekyoung Bae. 2020. HiFi-GAN: Generative Adversarial Networks for Efficient and High Fidelity Speech Synthesis. In NeurIPS."},{"key":"e_1_3_3_1_21_2","volume-title":"ICLR","author":"Lee Sang-gil","year":"2023","unstructured":"Sang-gil Lee, Wei Ping, Boris Ginsburg, Bryan Catanzaro, and Sungroh Yoon. 2023. BigVGAN: A Universal Neural Vocoder with Large-Scale Training. In ICLR."},{"key":"e_1_3_3_1_22_2","unstructured":"Yejin Lee Jaehoon Kang and Kyuhong Shim. 2025. P2VA: Converting Persona Descriptions into Voice Attributes for Fair and Controllable Text-to-Speech. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2505.17093 (2025)."},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10448268"},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"crossref","unstructured":"Yi Lei Shan Yang Xinsheng Wang and Lei Xie. 2022. MsEmoTTS: Multi-Scale Emotion Transfer Prediction and Control for Emotional Speech Synthesis. IEEE\/ACM Transactions on Audio Speech and Language Processing 30 (2022) 853\u2013864.","DOI":"10.1109\/TASLP.2022.3145293"},{"key":"e_1_3_3_1_25_2","unstructured":"Yingtie Lei Fanghai Yi Yihang Dong Weihuang Liu Xiaofeng Zhang Zimeng Li Chi-Man Pun and Xuhang Chen. 2025. Cmamrnet: A contextual mask-aware network enhancing mural restoration through comprehensive mask guidance. arXiv (2025)."},{"key":"e_1_3_3_1_26_2","doi-asserted-by":"publisher","DOI":"10.1145\/3746027.3754506"},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"crossref","unstructured":"Haolun Li and Chi-Man Pun. 2022. Monocular robust 3d human localization by global and body-parts depth awareness. IEEE Transactions on Circuits and Systems for Video Technology 32 11 (2022) 7692\u20137705.","DOI":"10.1109\/TCSVT.2022.3180737"},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i1.25214"},{"key":"e_1_3_3_1_29_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10890436"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"crossref","unstructured":"Xiaohong Li Guoheng Huang Lianglun Cheng Guo Zhong Weihuang Liu Xuhang Chen and Muyan Cai. 2024. Cross-domain visual prompting with spatial proximity knowledge distillation for histological image classification. Journal of Biomedical Informatics 158 (2024) 104728.","DOI":"10.1016\/j.jbi.2024.104728"},{"key":"e_1_3_3_1_31_2","volume-title":"NeurIPS","author":"Li Yinghao\u00a0Aaron","year":"2023","unstructured":"Yinghao\u00a0Aaron Li, Cong Han, Vinay\u00a0S. Raghavan, Gavin Mischler, and Nima Mesgarani. 2023. StyleTTS 2: Towards Human-Level Text-to-Speech through Style Diffusion and Adversarial Training with Large Speech Language Models. In NeurIPS."},{"key":"e_1_3_3_1_32_2","first-page":"21450","volume-title":"ICML","author":"Liu Haohe","year":"2023","unstructured":"Haohe Liu, Zehua Chen, Yi Yuan, Xinhao Mei, Xubo Liu, Danilo\u00a0P. Mandic, Wenwu Wang, and Mark\u00a0D. Plumbley. 2023. AudioLDM: Text-to-Audio Generation with Latent Diffusion Models. In ICML. 21450\u201321474."},{"key":"e_1_3_3_1_33_2","doi-asserted-by":"crossref","unstructured":"Weihuang Liu Xiaodong Cun and Chi-Man Pun. 2024. DH-GAN: Image manipulation localization via a dual homology-aware generative adversarial network. PR (2024) 110658.","DOI":"10.1016\/j.patcog.2024.110658"},{"key":"e_1_3_3_1_34_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i2.25263"},{"key":"e_1_3_3_1_35_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01818"},{"key":"e_1_3_3_1_36_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01862"},{"key":"e_1_3_3_1_37_2","unstructured":"Weihuang Liu Xi Shen Chi-Man Pun and Xiaodong Cun. 2024. ForgeryTTT: Zero-Shot Image Manipulation Localization with Test-Time Training. arXiv (2024)."},{"key":"e_1_3_3_1_38_2","unstructured":"Weihuang Liu Xi Shen Chi-Man Pun and Xiaodong Cun. 2025. Explicit visual prompting for universal foreground segmentations. TPAMI (2025)."},{"key":"e_1_3_3_1_39_2","unstructured":"Liangsi Lu Xuhang Chen Minzhe Guo Shichu Li Jingchao Wang and Yang Shi. 2026. ChordEdit: One-Step Low-Energy Transport for Image Editing. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2602.19083 (2026)."},{"key":"e_1_3_3_1_40_2","first-page":"463","volume-title":"WWW","author":"Lu Liangsi","year":"2026","unstructured":"Liangsi Lu, Jingchao Wang, Zhaorong Dai, Hanqian Liu, and Yang Shi. 2026. Riemannian Liquid Spatio-Temporal Graph Network. In WWW. 463\u2013474."},{"key":"e_1_3_3_1_41_2","unstructured":"Ziqin Luo Yihao Quan Xiaofeng Zhang Xiaosong Yuan and Chen Shen. 2026. ART: Attention Replacement Technique to Improve Factuality in LLMs. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2604.06393 (2026)."},{"key":"e_1_3_3_1_42_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.931"},{"key":"e_1_3_3_1_43_2","first-page":"8599","volume-title":"ICML","author":"Popov Vadim","year":"2021","unstructured":"Vadim Popov, Ivan Vovk, Vladimir Gogoryan, Tasnima Sadekova, and Mikhail\u00a0A. Kudinov. 2021. Grad-TTS: A Diffusion Probabilistic Model for Text-to-Speech. In ICML. 8599\u20138608."},{"key":"e_1_3_3_1_44_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1410"},{"key":"e_1_3_3_1_45_2","volume-title":"ICLR","author":"Ren Yi","year":"2021","unstructured":"Yi Ren, Chenxu Hu, Xu Tan, Tao Qin, Sheng Zhao, Zhou Zhao, and Tie-Yan Liu. 2021. FastSpeech 2: Fast and High-Quality End-to-End Text to Speech. In ICLR."},{"key":"e_1_3_3_1_46_2","unstructured":"RVC-Boss. 2024. GPT-SoVITS: A Powerful Few-shot Voice Conversion and Text-to-Speech WebUI. https:\/\/github.com\/RVC-Boss\/GPT-SoVITS."},{"key":"e_1_3_3_1_47_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"e_1_3_3_1_48_2","unstructured":"Yang Shi Yifeng Xie Minzhe Guo Liangsi Lu Mingxuan Huang Jingchao Wang Zhihong Zhu Boyan Xu and Zhiqi Huang. 2026. MMErroR: A Benchmark for Erroneous Reasoning in Vision-Language Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2601.03331 (2026)."},{"key":"e_1_3_3_1_49_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10445922"},{"key":"e_1_3_3_1_50_2","unstructured":"Chengyi Wang Sanyuan Chen Yu Wu Ziqiang Zhang Long Zhou Shujie Liu Zhuo Chen Yanqing Liu Huaming Wang Jinyu Li Lei He Sheng Zhao and Furu Wei. 2023. Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2301.02111 (2023)."},{"key":"e_1_3_3_1_51_2","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681076"},{"key":"e_1_3_3_1_52_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095969"},{"key":"e_1_3_3_1_53_2","unstructured":"Xiaoyu Xu Yulan Pan Xiaosong Yuan Zhihong Shen Minghao Su Yuanhao Su and Xiaofeng Zhang. 2026. Reasoning Fails Where Step Flow Breaks. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2604.06695 (2026)."},{"key":"e_1_3_3_1_54_2","doi-asserted-by":"crossref","unstructured":"Xin Yan Jiucheng Xie Mengqi Liu Haolun Li and Hao Gao. 2024. Hierarchical local temporal network for 2d-to-3d human pose estimation. IEEE Internet of Things Journal 12 1 (2024) 869\u2013880.","DOI":"10.1109\/JIOT.2024.3470751"},{"key":"e_1_3_3_1_55_2","doi-asserted-by":"publisher","DOI":"10.1145\/3746027.3754829"},{"key":"e_1_3_3_1_56_2","doi-asserted-by":"crossref","unstructured":"Shubo Yang Haolun Li Chi-Man Pun Chun Du and Hao Gao. 2024. Adaptive spatial-temporal graph-mixer for human motion prediction. IEEE Signal Processing Letters 31 (2024) 1244\u20131248.","DOI":"10.1109\/LSP.2024.3392686"},{"key":"e_1_3_3_1_57_2","doi-asserted-by":"crossref","unstructured":"Chi Zhang Hao Jiang Weihuang Liu Junyi Li Shiming Tang Mario Juhas and Yang Zhang. 2022. Correction of out-of-focus microscopic images by deep learning. Computational and Structural Biotechnology Journal 20 (2022) 1957\u20131966.","DOI":"10.1016\/j.csbj.2022.04.003"},{"key":"e_1_3_3_1_58_2","doi-asserted-by":"crossref","unstructured":"Xiaofeng Zhang Xuhang Chen Chaochen Gu Shanying Zhu Kim-Fung Tsang and Xinping Guan. 2026. Memory augment is all you need for image restoration. IEEE Transactions on Consumer Electronics (2026).","DOI":"10.1109\/TCE.2026.3655769"},{"key":"e_1_3_3_1_59_2","doi-asserted-by":"crossref","unstructured":"Xiaofeng Zhang Yihao Quan Chaochen Gu Chen Shen Xiaosong Yuan Shaotian Yan Hao Cheng Kaijie Wu and Jieping Ye. 2025. Shallow Focus Deep Fixes: Enhancing Shallow Layers Vision Attention Sinks to Alleviate Hallucination in LVLMs. (2025) 3512\u20133534.","DOI":"10.18653\/v1\/2025.emnlp-main.174"},{"key":"e_1_3_3_1_60_2","unstructured":"Xiaofeng Zhang Chen Shen Xiaosong Yuan Shaotian Yan Liang Xie Wenxiao Wang Chaochen Gu Hao Tang and Jieping Ye. 2024. From Redundancy to Relevance: Enhancing Explainability in Multimodal Large Language Models. NAACL (2024)."},{"key":"e_1_3_3_1_61_2","doi-asserted-by":"crossref","unstructured":"Xiaofeng Zhang Zishan Xu Hao Tang Chaochen Gu Wei Chen and Abdulmotaleb El\u00a0Saddik. 2025. Wakeup-Darkness: When Multimodal Meets Unsupervised Low-light Image Enhancement. ACM Transactions on Multimedia Computing Communications and Applications (2025).","DOI":"10.1145\/3711929"},{"key":"e_1_3_3_1_62_2","doi-asserted-by":"crossref","unstructured":"Xiaofeng Zhang Fanshuo Zeng and Chaochen Gu. 2024. Simignore: Exploring and enhancing multimodal large model complex reasoning via similarity computation. Neural Networks (2024) 107059.","DOI":"10.1016\/j.neunet.2024.107059"},{"key":"e_1_3_3_1_63_2","doi-asserted-by":"crossref","unstructured":"Xiaofeng Zhang Fanshuo Zeng Yihao Quan Zheng Hui and Jiawei Yao. 2025. Enhancing Multimodal Large Language Models Complex Reason via Similarity Computation. AAAI (2025) 10203\u201310211.","DOI":"10.1609\/aaai.v39i10.33107"},{"key":"e_1_3_3_1_64_2","doi-asserted-by":"crossref","unstructured":"Xiaofeng Zhang Yuanchao Zhu Chaochen Gu Jiawei Cao Hao Cheng and Kaijie Wu. 2026. What drives attention sinks? A study of massive activations and rotational positional encoding in large vision\u2013language models. Information Processing & Management 63 2 (2026) 104431.","DOI":"10.1016\/j.ipm.2025.104431"},{"key":"e_1_3_3_1_65_2","volume-title":"ICLR","author":"Zhang Xiaofeng","year":"2026","unstructured":"Xiaofeng Zhang, Yuanchao Zhu, Chaochen Gu, Xiaosong Yuan, Qiyan Zhao, Jiawei Cao, Feilong Tang, Sinan Fan, Yaomin Shen, Chen Shen, et\u00a0al. 2026. Hallucination Begins Where Saliency Drops. In ICLR."},{"key":"e_1_3_3_1_66_2","volume-title":"ICLR","author":"Zhao Qiyan","year":"2026","unstructured":"Qiyan Zhao, Xiaofeng Zhang, Shuochen Chang, Qianyu Chen, Xiaosong Yuan, Xuhang Chen, Luoqi Liu, Jiajun Zhang, Xu-Yao Zhang, and Da-Han Wang. 2026. Context Tokens are Anchors: Understanding the Repetition Curse in dMLLMs from an Information Flow Perspective. In ICLR."},{"key":"e_1_3_3_1_67_2","doi-asserted-by":"publisher","DOI":"10.1145\/3746027.3755271"},{"key":"e_1_3_3_1_68_2","doi-asserted-by":"publisher","DOI":"10.1109\/BIBM62325.2024.10822736"},{"key":"e_1_3_3_1_69_2","doi-asserted-by":"crossref","unstructured":"Leyi Zhu Weihuang Liu Xinyi Chen Zimeng Li Xuhang Chen Zhen Wang and Chi-Man Pun. 2024. Test-Time Intensity Consistency Adaptation for Shadow Detection. arXiv (2024).","DOI":"10.1007\/978-981-96-6594-5_16"},{"key":"e_1_3_3_1_70_2","unstructured":"Haomin Zuo Yidi Li Luoxiao Yang and Xiaofeng Zhang. 2026. Diffusion-CAM: Faithful Visual Explanations for dMLLMs. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2604.11005 (2026)."}],"event":{"name":"ICMR '26: International Conference on Multimedia Retrieval","location":"Amsterdam The Netherlands","acronym":"ICMR '26","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 2026 International Conference on Multimedia Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T15:11:06Z","timestamp":1781536266000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805622.3810570"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,15]]},"references-count":69,"alternative-id":["10.1145\/3805622.3810570","10.1145\/3805622"],"URL":"https:\/\/doi.org\/10.1145\/3805622.3810570","relation":{},"subject":[],"published":{"date-parts":[[2026,6,15]]},"assertion":[{"value":"2026-06-15","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}