{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,7]],"date-time":"2026-08-07T08:25:51Z","timestamp":1786091151807,"version":"3.56.0"},"reference-count":49,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2025,5,22]],"date-time":"2025-05-22T00:00:00Z","timestamp":1747872000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,5,22]],"date-time":"2025-05-22T00:00:00Z","timestamp":1747872000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Sci. China Inf. Sci."],"published-print":{"date-parts":[[2025,6]]},"DOI":"10.1007\/s11432-024-4401-y","type":"journal-article","created":{"date-parts":[[2025,5,27]],"date-time":"2025-05-27T04:05:08Z","timestamp":1748318708000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["MindScore: quantifying human preference for text-to-image generation through multi-view lens"],"prefix":"10.1007","volume":"68","author":[{"given":"Yiqi","family":"Tong","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiarui","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shaohang","family":"Wei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wei","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fuzhen","family":"Zhuang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Deqing","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xi","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Richeng","family":"Xuan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,5,22]]},"reference":[{"key":"4401_CR1","doi-asserted-by":"publisher","first-page":"2212","DOI":"10.1109\/TPAMI.2024.3522305","volume":"47","author":"F Bie","year":"2025","unstructured":"Bie F, Yang Y, Zhou Z, et al. RenAIssance: a survey into AI text-to-image generation in the era of large model. IEEE Trans Pattern Anal Mach Intell, 2025, 47: 2212\u20132231","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"4401_CR2","first-page":"8821","volume-title":"Proceedings of International Conference on Machine Learning","author":"A Ramesh","year":"2021","unstructured":"Ramesh A, Pavlov M, Goh G, et al. Zero-shot text-to-image generation. In: Proceedings of International Conference on Machine Learning, 2021. 8821\u20138831"},{"key":"4401_CR3","first-page":"10684","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"R Rombach","year":"2022","unstructured":"Rombach R, Blattmann A, Lorenz D, et al. High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022. 10684\u201310695"},{"key":"4401_CR4","first-page":"36479","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"C Saharia","year":"2022","unstructured":"Saharia C, Chan W, Saxena S, et al. Photorealistic text-to-image diffusion models with deep language understanding. In: Proceedings of Advances in Neural Information Processing Systems, 2022. 36479\u201336494"},{"key":"4401_CR5","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"T Lee","year":"2024","unstructured":"Lee T, Yasunaga M, Meng C L, et al. Holistic valuation of text-to-image models. In: Proceedings of Advances in Neural Information Processing Systems, 2024"},{"key":"4401_CR6","first-page":"19401","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Y W Liang","year":"2024","unstructured":"Liang Y W, He J F, Li G, et al. Rich human feedback for text-to-image generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024. 19401\u201319411"},{"key":"4401_CR7","first-page":"28140","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"A Sarkar","year":"2024","unstructured":"Sarkar A, Mai H L, Mahapatra A, et al. Shadows don\u2019t lie and lines can\u2019t bend! Generative models don\u2019t know projective geometry\u2026 for now. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024. 28140\u201328149"},{"key":"4401_CR8","volume-title":"Evaluating text to image synthesis: survey and taxonomy of image quality metrics","author":"S Hartwig","year":"2024","unstructured":"Hartwig S, Engel D, Sick L, et al. Evaluating text to image synthesis: survey and taxonomy of image quality metrics. 2024. ArXiv:2403.11821"},{"key":"4401_CR9","doi-asserted-by":"publisher","first-page":"220101","DOI":"10.1007\/s11432-024-4231-5","volume":"67","author":"Z Chen","year":"2024","unstructured":"Chen Z, Wang W Y, Tian H, et al. How far are we to GPT-4V? Closing the gap to commercial multimodal models with open-source suites. Sci China Inf Sci, 2024, 67: 220101","journal-title":"Sci China Inf Sci"},{"key":"4401_CR10","doi-asserted-by":"publisher","first-page":"101832","DOI":"10.1016\/j.intell.2024.101832","volume":"104","author":"G E Gignac","year":"2024","unstructured":"Gignac G E, Szodorai E T. Defining intelligence: bridging the gap between human and artificial perspectives. Intelligence, 2024, 104: 101832","journal-title":"Intelligence"},{"key":"4401_CR11","doi-asserted-by":"publisher","first-page":"489","DOI":"10.1348\/0007126042369811","volume":"95","author":"H Leder","year":"2004","unstructured":"Leder H, Belke B, Oeberst A, et al. A model of aesthetic appreciation and aesthetic judgments. Br J Psychol, 2004, 95: 489\u2013508","journal-title":"Br J Psychol"},{"key":"4401_CR12","first-page":"189","volume-title":"Aesthetic Science: Connecting Minds, Brains, and Experience. Oxford: Oxford University Press","author":"S E Palmer","year":"2012","unstructured":"Palmer S E, Schloss K B, Sammartino J. Hidden knowledge in aesthetic judgments: preferences for color and spatial composition. In: Aesthetic Science: Connecting Minds, Brains, and Experience. Oxford: Oxford University Press, 2012. 189\u2013222"},{"key":"4401_CR13","doi-asserted-by":"publisher","first-page":"77","DOI":"10.1146\/annurev-psych-120710-100504","volume":"64","author":"S E Palmer","year":"2013","unstructured":"Palmer S E, Schloss K B, Sammartino J. Visual aesthetics and human preference. Annu Rev Psychol, 2013, 64: 77\u2013107","journal-title":"Annu Rev Psychol"},{"key":"4401_CR14","first-page":"1941","volume-title":"Proceedings of the ACM Designing Interactive Systems Conference","author":"L Y Chiou","year":"2023","unstructured":"Chiou L Y, Hung P K, Liang R H, et al. Designing with AI: an exploration of co-ideation with image generators. In: Proceedings of the ACM Designing Interactive Systems Conference, 2023. 1941\u20131954"},{"key":"4401_CR15","first-page":"54","volume":"8","author":"M Sutrop","year":"2020","unstructured":"Sutrop M. Challenges of aligning artificial intelligence with human values. Acta Baltica Hist et Phil Sci, 2020, 8: 54\u201372","journal-title":"Acta Baltica Hist et Phil Sci"},{"key":"4401_CR16","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"Y Kirstain","year":"2024","unstructured":"Kirstain Y, Polyak A, Singer U, et al. Pick-a-Pic: an open dataset of user preferences for text-to-image generation. In: Proceedings of Advances in Neural Information Processing Systems, 2024"},{"key":"4401_CR17","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"J Z Xu","year":"2024","unstructured":"Xu J Z, Liu X, Wu Y C, et al. ImageReward: learning and evaluating human preferences for text-to-image generation. In: Proceedings of Advances in Neural Information Processing Systems, 2024"},{"key":"4401_CR18","volume-title":"Proceedings of the 41st International Conference on Machine Learning","author":"P Esser","year":"2024","unstructured":"Esser P, Kulal S, Blattmann A, et al. Scaling rectified flow transformers for high-resolution image synthesis. In: Proceedings of the 41st International Conference on Machine Learning, 2024"},{"key":"4401_CR19","first-page":"2096","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"X S Wu","year":"2023","unstructured":"Wu X S, Sun K Q, Zhu F, et al. Human preference score: better aligning text-to-image models with human preference. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023. 2096\u20132105"},{"key":"4401_CR20","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"T Salimans","year":"2016","unstructured":"Salimans T, Goodfellow I, Zaremba W, et al. Improved techniques for training GANs. In: Proceedings of Advances in Neural Information Processing Systems, 2016"},{"key":"4401_CR21","first-page":"1316","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"T Xu","year":"2018","unstructured":"Xu T, Zhang P C, Huang Q Y, et al. AttnGAN: fine-grained text to image generation with attentional generative adversarial networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018. 1316\u20131324"},{"key":"4401_CR22","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"M Heusel","year":"2017","unstructured":"Heusel M, Ramsauer H, Unterthiner T, et al. GANs trained by a two time-scale update rule converge to a local Nash equilibrium. In: Proceedings of Advances in Neural Information Processing Systems, 2017"},{"key":"4401_CR23","first-page":"594","volume-title":"Proceedings of European Conference on Computer Vision","author":"T M Dinh","year":"2022","unstructured":"Dinh T M, Nguyen R, Hua B S. TISE: bag of metrics for text-to-image synthesis evaluation. In: Proceedings of European Conference on Computer Vision, 2022. 594\u2013609"},{"key":"4401_CR24","doi-asserted-by":"publisher","first-page":"109636","DOI":"10.1016\/j.patcog.2023.109636","volume":"141","author":"Q Liu","year":"2023","unstructured":"Liu Q, He X, Teng Q, et al. BDNet: a BERT-based dual-path network for text-to-image cross-modal person re-identification. Pattern Recogn, 2023, 141: 109636","journal-title":"Pattern Recogn"},{"key":"4401_CR25","first-page":"7514","volume-title":"Proceedings of the Conference on Empirical Methods in Natural Language Processing","author":"J Hessel","year":"2021","unstructured":"Hessel J, Holtzman A, Forbes M, et al. CLIPScore: a reference-free evaluation metric for image captioning. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing, 2021. 7514\u20137528"},{"key":"4401_CR26","first-page":"8748","volume-title":"Proceedings of International Conference on Machine Learning","author":"A Radford","year":"2021","unstructured":"Radford A, Kim J W, Hallacy C, et al. Learning transferable visual models from natural language supervision. In: Proceedings of International Conference on Machine Learning, 2021. 8748\u20138763"},{"key":"4401_CR27","first-page":"2555","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","author":"J Y Wang","year":"2023","unstructured":"Wang J Y, Chan K C, Loy C C. Exploring clip for assessing the look and feel of images. In: Proceedings of the AAAI Conference on Artificial Intelligence, 2023. 2555\u20132563"},{"key":"4401_CR28","first-page":"12888","volume-title":"Proceedings of International Conference on Machine Learning","author":"J N Li","year":"2022","unstructured":"Li J N, Li D X, Xiong C M, et al. BLIP: bootstrapping language-image pre-training for unified vision-language understanding and generation. In: Proceedings of International Conference on Machine Learning, 2022. 12888\u201312900"},{"key":"4401_CR29","first-page":"19730","volume-title":"Proceedings of International Conference on Machine Learning","author":"J N Li","year":"2023","unstructured":"Li J N, Li D X, Savarese S, et al. BLIP-2: bootstrapping language-image pre-training with frozen image encoders and large language models. In: Proceedings of International Conference on Machine Learning, 2023. 19730\u201319742"},{"key":"4401_CR30","first-page":"20406","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"Y S Hu","year":"2023","unstructured":"Hu Y S, Liu B L, Kasai J, et al. TIFA: accurate and interpretable text-to-image faithfulness evaluation with question answering. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023. 20406\u201320417"},{"key":"4401_CR31","first-page":"70799","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"J Singh","year":"2023","unstructured":"Singh J, Zheng L. Divide, evaluate, and refine: evaluating and improving text-to-image alignment with iterative VQA feedback. In: Proceedings of Advances in Neural Information Processing Systems, 2023. 70799\u201370811"},{"key":"4401_CR32","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"M Yarom","year":"2024","unstructured":"Yarom M, Bitton Y, Changpinyo S, et al. What you see is what you read? Improving text-image alignment evaluation. In: Proceedings of Advances in Neural Information Processing Systems, 2024"},{"key":"4401_CR33","doi-asserted-by":"publisher","first-page":"422","DOI":"10.1038\/20833","volume":"399","author":"R P Taylor","year":"1999","unstructured":"Taylor R P, Micolich A P, Jonas D. Fractal analysis of Pollock\u2019s drip paintings. Nature, 1999, 399: 422","journal-title":"Nature"},{"key":"4401_CR34","first-page":"770","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"K M He","year":"2016","unstructured":"He K M, Zhang X Y, Ren S Q, et al. Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016. 770\u2013778"},{"key":"4401_CR35","volume-title":"An image is worth 16x16 words: transformers for image recognition at scale","author":"A Dosovitskiy","year":"2020","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A, et al. An image is worth 16x16 words: transformers for image recognition at scale. 2020. ArXiv:2010.11929"},{"key":"4401_CR36","first-page":"1191","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"J T Lee","year":"2019","unstructured":"Lee J T, Kim C S. Image aesthetic assessment based on pairwise comparison a unified approach to score regression, binary classification, and personalization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2019. 1191\u20131200"},{"key":"4401_CR37","volume-title":"Proceedings of the 23rd International Workshop on Multimedia Signal Processing (MMSP)","author":"Z Y Lei","year":"2021","unstructured":"Lei Z Y, Xie Y J, Ling S Y, et al. Multi-modal aesthetic assessment for mobile gaming image. In: Proceedings of the 23rd International Workshop on Multimedia Signal Processing (MMSP), 2021"},{"key":"4401_CR38","volume-title":"Proceedings of International Conference on Visual Communications and Image Processing (VCIP)","author":"T Wang","year":"2021","unstructured":"Wang T, Sun W, Min X K, et al. A multi-dimensional aesthetic quality assessment model for mobile game images. In: Proceedings of International Conference on Visual Communications and Image Processing (VCIP), 2021"},{"key":"4401_CR39","first-page":"24480","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"U Ojha","year":"2023","unstructured":"Ojha U, Li Y H, Lee Y J. Towards universal fake image detectors that generalize across generative models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023. 24480\u201324489"},{"key":"4401_CR40","first-page":"22445","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"Z D Wang","year":"2023","unstructured":"Wang Z D, Bao J M, Zhou W G, et al. Dire for diffusion-generated image detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023. 22445\u201322455"},{"key":"4401_CR41","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"M J Zhu","year":"2024","unstructured":"Zhu M J, Chen H T, Yan Q Y, et al. GenImage: a million-scale benchmark for detecting AI-generated image. In: Proceedings of Advances in Neural Information Processing Systems, 2024"},{"key":"4401_CR42","first-page":"8018","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"S X Zhang","year":"2024","unstructured":"Zhang S X, Wang B H, Wu J Q, et al. Learning multi-dimensional human preference for text-to-image generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024. 8018\u20138027"},{"key":"4401_CR43","first-page":"1597","volume-title":"Proceedings of International Conference on Machine Learning","author":"T Chen","year":"2020","unstructured":"Chen T, Kornblith S, Norouzi M, et al. A simple framework for contrastive learning of visual representations. In: Proceedings of International Conference on Machine Learning, 2020. 1597\u20131607"},{"key":"4401_CR44","doi-asserted-by":"publisher","first-page":"121101","DOI":"10.1007\/s11432-024-4222-0","volume":"68","author":"Z H Xi","year":"2025","unstructured":"Xi Z H, Chen W X, Guo X, et al. The rise and potential of large language model based agents: a survey. Sci China Inf Sci, 2025, 68: 121101","journal-title":"Sci China Inf Sci"},{"key":"4401_CR45","volume-title":"Qwen-VL: a frontier large vision-language model with versatile abilities","author":"J Z Bai","year":"2023","unstructured":"Bai J Z, Bai S, Yang S S, et al. Qwen-VL: a frontier large vision-language model with versatile abilities. 2023. ArXiv:2308.12966"},{"key":"4401_CR46","doi-asserted-by":"publisher","DOI":"10.36227\/techrxiv.170723324.44685515\/v1","volume-title":"Detecting multimedia generated by large AI models: a survey","author":"L Lin","year":"2024","unstructured":"Lin L, Gupta N, Zhang Y, et al. Detecting multimedia generated by large AI models: a survey. 2024. ArXiv:2402.00045"},{"key":"4401_CR47","first-page":"25278","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"C Schuhmann","year":"2022","unstructured":"Schuhmann C, Beaumont R, Vencu R, et al. LAION-5B: an open large-scale dataset for training next generation image-text models. In: Proceedings of Advances in Neural Information Processing Systems, 2022. 25278\u201325294"},{"key":"4401_CR48","volume-title":"Decoupled weight decay regularization","author":"I Loshchilov","year":"2017","unstructured":"Loshchilov I, Hutter F. Decoupled weight decay regularization. 2017. ArXiv:1711.05101"},{"key":"4401_CR49","first-page":"2579","volume":"9","author":"L van der Maaten","year":"2008","unstructured":"van der Maaten L, Hinton G. Visualizing data using t-SNE. J Mach Learn Res, 2008, 9: 2579\u20132605","journal-title":"J Mach Learn Res"}],"container-title":["Science China Information Sciences"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-024-4401-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11432-024-4401-y","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-024-4401-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T21:03:25Z","timestamp":1784581405000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11432-024-4401-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,22]]},"references-count":49,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2025,6]]}},"alternative-id":["4401"],"URL":"https:\/\/doi.org\/10.1007\/s11432-024-4401-y","relation":{},"ISSN":["1674-733X","1869-1919"],"issn-type":[{"value":"1674-733X","type":"print"},{"value":"1869-1919","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,5,22]]},"assertion":[{"value":"14 November 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 February 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 April 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 May 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"160105"}}