{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T15:45:17Z","timestamp":1781797517090,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":67,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755379","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:54:15Z","timestamp":1761375255000},"page":"6948-6957","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Multi-Dimensional Text-to-Face Image Quality Assessment Using LLM: Database and Method"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6292-0529","authenticated-orcid":false,"given":"Yixuan","family":"Gao","sequence":"first","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5693-0416","authenticated-orcid":false,"given":"Xiongkuo","family":"Min","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-8941-2981","authenticated-orcid":false,"given":"Jinliang","family":"Han","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5087-6559","authenticated-orcid":false,"given":"Yuqin","family":"Cao","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-7753-1596","authenticated-orcid":false,"given":"Sijing","family":"Wu","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-9866-218X","authenticated-orcid":false,"given":"Yunze","family":"Dou","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8165-9322","authenticated-orcid":false,"given":"Guangtao","family":"Zhai","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2022.108246"},{"key":"e_1_3_2_1_2_1","volume-title":"Methodology for the subjective assessment of the quality of television pictures","author":"Recommendation ITU-R BT.","year":"2002","unstructured":"Recommendation ITU-R BT. 2002. Methodology for the subjective assessment of the quality of television pictures. International Telecommunication Union (2002)."},{"key":"e_1_3_2_1_3_1","volume-title":"European Conference on Computer Vision. Springer, 204-222","author":"Chatterjee Agneet","year":"2024","unstructured":"Agneet Chatterjee, Gabriela Ben Melech Stan, Estelle Aflalo, Sayak Paul, Dhruba Ghosh, Tejas Gokhale, Ludwig Schmidt, Hannaneh Hajishirzi, Vasudev Lal, Chitta Baral, et al. 2024. Getting it right: Improving spatial consistency in text-to-image models. In European Conference on Computer Vision. Springer, 204-222."},{"key":"e_1_3_2_1_4_1","volume-title":"Topiq: A top-down approach from semantics to distortions for image quality assessment","author":"Chen Chaofeng","year":"2024","unstructured":"Chaofeng Chen, Jiadi Mo, Jingwen Hou, Haoning Wu, Liang Liao, Wenxiu Sun, Qiong Yan, and Weisi Lin. 2024. Topiq: A top-down approach from semantics to distortions for image quality assessment. IEEE Transactions on Image Processing (2024)."},{"key":"e_1_3_2_1_5_1","unstructured":"Junsong Chen Jincheng Yu Chongjian Ge Lewei Yao Enze Xie Yue Wu ZhongdaoWang James Kwok Ping Luo Huchuan Lu et al. 2023. Pixart-?: Fast training of diffusion transformer for photorealistic text-to-image synthesis. arXiv preprint arXiv:2310.00426 (2023)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00283"},{"key":"e_1_3_2_1_7_1","unstructured":"Ming Ding Zhuoyi Yang Wenyi Hong Wendi Zheng Chang Zhou Da Yin Junyang Lin Xu Zou Zhou Shao Hongxia Yang and Jie Tang. 2021. CogView: Mastering Text-to-Image Generation via Transformers. arXiv:2105.13290 [cs.CV]"},{"key":"e_1_3_2_1_8_1","volume-title":"Image Quality Score Distribution Prediction via Alpha Stable Model","author":"Gao Yixuan","year":"2022","unstructured":"Yixuan Gao, Xiongkuo Min, Wenhan Zhu, Xiao-Ping Zhang, and Guangtao Zhai. 2022. Image Quality Score Distribution Prediction via Alpha Stable Model. IEEE Transactions on Circuits and Systems for Video Technology (2022)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2022.3229839"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547872"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2023.3295375"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV51458.2022.00404"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01043"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00568"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICB45273.2019.8987255"},{"key":"e_1_3_2_1_16_1","unstructured":"Wenyi Hong Weihan Wang Ming Ding Wenmeng Yu Qingsong Lv Yan Wang Yean Cheng Shiyu Huang Junhui Ji Zhao Xue et al. 2024. Cogvlm2: Visual language models for image and video understanding. arXiv preprint arXiv:2408.16500 (2024)."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2020.2967829"},{"key":"e_1_3_2_1_18_1","volume-title":"T2I-CompBench: An Enhanced and Comprehensive Benchmark for Compositional Text-to-Image Generation","author":"Huang Kaiyi","year":"2025","unstructured":"Kaiyi Huang, Chengqi Duan, Kaiyue Sun, Enze Xie, Zhenguo Li, and Xihui Liu. 2025. T2I-CompBench: An Enhanced and Comprehensive Benchmark for Compositional Text-to-Image Generation. IEEE Transactions on Pattern Analysis and Machine Intelligence (2025)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00589"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.224"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1177\/0301006619869134"},{"key":"e_1_3_2_1_22_1","volume-title":"MEBeauty: a multi-ethnic facial beauty dataset in-the-wild. Neural Computing and Applications","author":"Lebedeva Irina","year":"2022","unstructured":"Irina Lebedeva, Yi Guo, and Fangli Ying. 2022. MEBeauty: a multi-ethnic facial beauty dataset in-the-wild. Neural Computing and Applications (2022), 1-15."},{"key":"e_1_3_2_1_23_1","volume-title":"Controllable text-to-image generation. Advances in Neural Information Processing Systems 32","author":"Li Bowen","year":"2019","unstructured":"Bowen Li, Xiaojuan Qi, Thomas Lukasiewicz, and Philip Torr. 2019. Controllable text-to-image generation. Advances in Neural Information Processing Systems 32 (2019)."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2023.3319020"},{"key":"e_1_3_2_1_25_1","volume-title":"Llavanext: Improved reasoning, ocr, and world knowledge.","author":"Liu Haotian","year":"2024","unstructured":"Haotian Liu, Chunyuan Li, Yuheng Li, Bo Li, Yuanhan Zhang, Sheng Shen, and Yong Jae Lee. 2024. Llavanext: Improved reasoning, ocr, and world knowledge."},{"key":"e_1_3_2_1_26_1","volume-title":"F-Bench: Rethinking Human Preference Evaluation Metrics for Benchmarking Face Generation, Customization, and Restoration. arXiv preprint arXiv:2412.13155","author":"Liu Lu","year":"2024","unstructured":"Lu Liu, Huiyu Duan, Qiang Hu, Liu Yang, Chunlei Cai, Tianxiao Ye, Huayu Liu, Xiaoyun Zhang, and Guangtao Zhai. 2024. F-Bench: Rethinking Human Preference Evaluation Metrics for Benchmarking Face Generation, Customization, and Restoration. arXiv preprint arXiv:2412.13155 (2024)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCDS.2025.3529177"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2017.2729020"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.425"},{"key":"e_1_3_2_1_30_1","unstructured":"Haoyu Lu Wen Liu Bo Zhang BingxuanWang Kai Dong Bo Liu Jingxiang Sun Tongzheng Ren Zhuoshu Li Hao Yang et al. 2024. Deepseek-vl: towards realworld vision-language understanding. arXiv preprint arXiv:2403.05525 (2024)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01400"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503250"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2025.3544659"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2012.2214050"},{"key":"e_1_3_2_1_35_1","volume-title":"Facexformer: A unified transformer for facial analysis. arXiv preprint arXiv:2403.12960","author":"Narayan Kartik","year":"2024","unstructured":"Kartik Narayan, Vibashan VS, Rama Chellappa, and Vishal M Patel. 2024. Facexformer: A unified transformer for facial analysis. arXiv preprint arXiv:2403.12960 (2024)."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2022.3188991"},{"key":"e_1_3_2_1_37_1","volume-title":"Sdxl: Improving latent diffusion models for high-resolution image synthesis. arXiv preprint arXiv:2307.01952","author":"Podell Dustin","year":"2023","unstructured":"Dustin Podell, Zion English, Kyle Lacey, Andreas Blattmann, Tim Dockhorn, Jonas M\u00fcller, Joe Penna, and Robin Rombach. 2023. Sdxl: Improving latent diffusion models for high-resolution image synthesis. arXiv preprint arXiv:2307.01952 (2023)."},{"key":"e_1_3_2_1_38_1","volume-title":"International conference on machine learning. Pmlr, 8821-8831","author":"Ramesh Aditya","year":"2021","unstructured":"Aditya Ramesh, Mikhail Pavlov, Gabriel Goh, Scott Gray, Chelsea Voss, Alec Radford, Mark Chen, and Ilya Sutskever. 2021. Zero-shot text-to-image generation. In International conference on machine learning. Pmlr, 8821-8831."},{"key":"e_1_3_2_1_39_1","volume-title":"Kandinsky: an improved text-to-image synthesis with image prior and latent diffusion. arXiv preprint arXiv:2310.03502","author":"Razzhigaev Anton","year":"2023","unstructured":"Anton Razzhigaev, Arseniy Shakhmatov, Anastasia Maltseva, Vladimir Arkhipkin, Igor Pavlov, Ilya Ryabov, Angelina Kuts, Alexander Panchenko, Andrey Kuznetsov, and Denis Dimitrov. 2023. Kandinsky: an improved text-to-image synthesis with image prior and latent diffusion. arXiv preprint arXiv:2310.03502 (2023)."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2012.2197011"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"crossref","unstructured":"Robin Rombach Andreas Blattmann Dominik Lorenz Patrick Esser and Bj\u00f6rn Ommer. 2021. High-Resolution Image Synthesis with Latent Diffusion Models. arXiv:2112.10752 [cs.CV]","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3507901"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2006.881959"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2023.3301276"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00372"},{"key":"e_1_3_2_1_46_1","volume-title":"AnyFace: A unified framework for free-style text-to-face synthesis and manipulation","author":"Sun Jianxin","year":"2024","unstructured":"Jianxin Sun, Qiyao Deng, Qi Li, Muyi Sun, Yunfan Liu, and Zhenan Sun. 2024. AnyFace: A unified framework for free-style text-to-face synthesis and manipulation. IEEE Transactions on Pattern Analysis and Machine Intelligence (2024)."},{"key":"e_1_3_2_1_47_1","volume-title":"GraphIQA: Learning distortion graph representations for blind image quality assessment","author":"Sun Simeng","year":"2022","unstructured":"Simeng Sun, Tao Yu, Jiahua Xu, Wei Zhou, and Zhibo Chen. 2022. GraphIQA: Learning distortion graph representations for blind image quality assessment. IEEE Transactions on Multimedia (2022)."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2023.3270621"},{"key":"e_1_3_2_1_49_1","volume-title":"CLIP-AGIQA: Boosting the Performance of AI-Generated Image Quality Assessment with CLIP. In International Conference on Pattern Recognition. Springer, 48-61","author":"Tang Zhenchen","year":"2025","unstructured":"Zhenchen Tang, Zichuan Wang, Bo Peng, and Jing Dong. 2025. CLIP-AGIQA: Boosting the Performance of AI-Generated Image Quality Assessment with CLIP. In International Conference on Pattern Recognition. Springer, 48-61."},{"key":"e_1_3_2_1_50_1","volume-title":"Internlm: A multilingual language model with progressively enhanced capabilities.","author":"Team LM","year":"2023","unstructured":"InternLM Team. 2023. Internlm: A multilingual language model with progressively enhanced capabilities."},{"key":"e_1_3_2_1_51_1","volume-title":"Generalized visual quality assessment of GAN-generated face images. arXiv preprint arXiv:2201.11975","author":"Tian Yu","year":"2022","unstructured":"Yu Tian, Zhangkai Ni, Baoliang Chen, Shiqi Wang, Hanli Wang, and Sam Kwong. 2022. Generalized visual quality assessment of GAN-generated face images. arXiv preprint arXiv:2201.11975 (2022)."},{"key":"e_1_3_2_1_52_1","volume-title":"CAAI International Conference on Artificial Intelligence. Springer, 46-57","author":"Duan Huiyu","year":"2023","unstructured":"JiaruiWang, Huiyu Duan, Jing Liu, Shi Chen, Xiongkuo Min, and Guangtao Zhai. 2023. Aigciqa2023: A large-scale image quality assessment database for ai generated images: from the perspectives of quality, authenticity and correspondence. In CAAI International Conference on Artificial Intelligence. Springer, 46-57."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2023.3243683"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"crossref","unstructured":"Zhou Wang and Alan Conrad Bovik. 2006. Modern image quality assessment. Ph.D. Dissertation. Springer.","DOI":"10.1007\/978-3-031-02238-8"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2003.819861"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3680939"},{"key":"e_1_3_2_1_57_1","unstructured":"Zhiyu Wu Xiaokang Chen Zizheng Pan Xingchao Liu Wen Liu Damai Dai Huazuo Gao Yiyang Ma Chengyue Wu Bingxuan Wang et al. 2024. Deepseekvl2: Mixture-of-experts vision-language models for advanced multimodal understanding. arXiv preprint arXiv:2412.10302 (2024)."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00229"},{"key":"e_1_3_2_1_59_1","volume-title":"Comboloss for facial attractiveness analysis with squeeze-and-excitation networks. arXiv preprint arXiv:2010.10721","author":"Xu Lu","year":"2020","unstructured":"Lu Xu and Jinhai Xiang. 2020. Comboloss for facial attractiveness analysis with squeeze-and-excitation networks. arXiv preprint arXiv:2010.10721 (2020)."},{"key":"e_1_3_2_1_60_1","unstructured":"An Yang Baosong Yang Binyuan Hui Bo Zheng Bowen Yu Chang Zhou Chengpeng Li Chengyuan Li Dayiheng Liu Fei Huang et al. 2024. Qwen2 technical report. arXiv preprint arXiv:2407.10671 (2024)."},{"key":"e_1_3_2_1_61_1","volume-title":"Proceedings of the ieee\/cvf conference on computer vision and pattern recognition. 13040-13051","author":"Ye Qinghao","year":"2024","unstructured":"Qinghao Ye, Haiyang Xu, Jiabo Ye, Ming Yan, Anwen Hu, Haowei Liu, Qi Qian, Ji Zhang, and Fei Huang. 2024. mplug-owl2: Revolutionizing multi-modal large language model with modality collaboration. In Proceedings of the ieee\/cvf conference on computer vision and pattern recognition. 13040-13051."},{"key":"e_1_3_2_1_62_1","volume-title":"European Conference on Computer Vision. Springer, 259-276","author":"You Zhiyuan","year":"2024","unstructured":"Zhiyuan You, Zheyuan Li, Jinjin Gu, Zhenfei Yin, Tianfan Xue, and Chao Dong. 2024. Depicting beyond scores: Advancing image quality assessment through multi-modal language models. In European Conference on Computer Vision. Springer, 259-276."},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-019-2757-1"},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01100"},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP.2012.6467150"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2018.2886771"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00877"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755379","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:00:54Z","timestamp":1765339254000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755379"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":67,"alternative-id":["10.1145\/3746027.3755379","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755379","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}