{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T09:09:23Z","timestamp":1765357763847,"version":"3.38.0"},"reference-count":31,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,1,8]],"date-time":"2025-01-08T00:00:00Z","timestamp":1736294400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,8]],"date-time":"2025-01-08T00:00:00Z","timestamp":1736294400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100012174","name":"Hunan Office of Philosophy and Social Science","doi-asserted-by":"publisher","award":["23YBQ046"],"award-info":[{"award-number":["23YBQ046"]}],"id":[{"id":"10.13039\/501100012174","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2025,2]]},"DOI":"10.1007\/s00530-024-01610-7","type":"journal-article","created":{"date-parts":[[2025,1,8]],"date-time":"2025-01-08T06:14:44Z","timestamp":1736316884000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Psychological analysis of house-tree-person drawings based on multimodal large models"],"prefix":"10.1007","volume":"31","author":[{"given":"Dahong","family":"Xu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Siyu","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yihan","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xi","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,1,8]]},"reference":[{"key":"1610_CR1","doi-asserted-by":"publisher","first-page":"1778","DOI":"10.1186\/s12889-022-13943-x","volume":"22","author":"F Campbell","year":"2022","unstructured":"Campbell, F., Blank, L., Cantrell, A., et al.: Factors that influence mental health of university and college students in the UK: a systematic review. BMC Public Health 22, 1778 (2022). https:\/\/doi.org\/10.1186\/s12889-022-13943-x","journal-title":"BMC Public Health"},{"key":"1610_CR2","doi-asserted-by":"publisher","first-page":"100636","DOI":"10.1016\/j.xjep.2023.100636","volume":"32","author":"ES Al-Rasheed","year":"2023","unstructured":"Al-Rasheed, E.S., Al-Rasheed, M.S.: The value ofpainting as a therapeutic tool in the treatment of anxiety\/depression mental disorders. J Interprof Educ Pract. 32, 100636 (2023). https:\/\/doi.org\/10.1016\/j.xjep.2023.100636","journal-title":"J Interprof Educ Pract."},{"key":"1610_CR3","doi-asserted-by":"publisher","first-page":"1041770","DOI":"10.3389\/fpsyt.2022.1041770","volume":"13","author":"H Guo","year":"2023","unstructured":"Guo, H., Feng, B., Ma, Y., Zhang, X., Fan, H., Dong, Z., Chen, T., Gong, Q.: Analysis of the screening and predicting characteristics of the house-tree-person drawing test for mental disorders: a systematic review and meta-analysis. Front Psychiatry. 13, 1041770 (2023). https:\/\/doi.org\/10.3389\/fpsyt.2022.1041770","journal-title":"Front Psychiatry."},{"issue":"3","key":"1610_CR4","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3617592","volume":"56","author":"T Ghandi","year":"2023","unstructured":"Ghandi, T., Pourreza, H., Mahyar, H.: Deep learning approaches on image captioning: a review. ACM Comput Surv. 56(3), 1\u201339 (2023). https:\/\/doi.org\/10.1145\/3617592","journal-title":"ACM Comput Surv."},{"issue":"19","key":"1610_CR5","doi-asserted-by":"publisher","first-page":"9331","DOI":"10.3390\/app131910894","volume":"12","author":"G Al-Sayyad","year":"2022","unstructured":"Al-Sayyad, G., Almotairi, S., Al-Badr, A.: A systematic literature review on using the encoder-decoder models for image captioning in English and Arabic languages. Appl Sci 12(19), 9331 (2022). https:\/\/doi.org\/10.3390\/app131910894","journal-title":"Appl Sci"},{"doi-asserted-by":"publisher","unstructured":"Li J, Li D, Savarese S, Hoi S (2023) Blip-2: Bootstrap language-image pre-training with frozen image encoders and large language models. In: International Conference on Machine Learning. pp 19730\u201319742. https:\/\/doi.org\/10.48550\/arXiv.2301.12597","key":"1610_CR6","DOI":"10.48550\/arXiv.2301.12597"},{"issue":"23","key":"1610_CR7","doi-asserted-by":"publisher","first-page":"1","DOI":"10.48550\/arXiv.1907.04433","volume":"21","author":"J Guo","year":"2020","unstructured":"Guo, J., He, H., He, T., Lausen, L., Li, M., Lin, H., Shi, X., Wang, C., Zha, S., Zhang, A.: Gluoncv and gluonnlp: deep learning in computer vision and natural language processing. J Mach Learn Res. 21(23), 1\u20137 (2020). https:\/\/doi.org\/10.48550\/arXiv.1907.04433","journal-title":"J Mach Learn Res."},{"key":"1610_CR8","doi-asserted-by":"publisher","DOI":"10.1016\/j.nlp.2023.100026","author":"W Khan","year":"2023","unstructured":"Khan, W., Daud, A., Khan, K., Muhammad, S., Haq, R.: Exploring the frontiers of deep learning and natural language processing: a comprehensive overview of key challenges and emerging trends. Nat Lang Proc J (2023). https:\/\/doi.org\/10.1016\/j.nlp.2023.100026","journal-title":"Nat Lang Proc J"},{"issue":"19","key":"1610_CR9","doi-asserted-by":"publisher","first-page":"10894","DOI":"10.3390\/app131910894","volume":"13","author":"A Alsayed","year":"2023","unstructured":"Alsayed, A., Arif, M., Qadah, T.M., Alotaibi, S.: Asystematic literature review on using the encoder-decoder models for image captioning in English and Arabic languages. Appl Sci 13(19), 10894 (2023). https:\/\/doi.org\/10.3390\/app131910894","journal-title":"Appl Sci"},{"doi-asserted-by":"publisher","unstructured":"Vinyals O, Toshev A, Bengio S, Erhan D (2015) Show and tell: a neural image caption generator. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 3156\u20133164. https:\/\/doi.org\/10.1109\/CVPR.2015.7298935","key":"1610_CR10","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"1610_CR11","doi-asserted-by":"publisher","first-page":"2048","DOI":"10.5555\/3045118.3045336","volume":"37","author":"K Xu","year":"2015","unstructured":"Xu, K., Ba, J., Kiros, R., Cho, K., Courville, A., Salakhudinov, R., Zemel, R., Bengio, Y.: Show, attend and tell: neural image caption generation with visual attention. Int Conf Mach Learn. 37, 2048\u20132057 (2015). https:\/\/doi.org\/10.5555\/3045118.3045336","journal-title":"Int Conf Mach Learn."},{"key":"1610_CR12","doi-asserted-by":"publisher","first-page":"335","DOI":"10.1007\/s00521-015-1846-7","volume":"27","author":"L Bai","year":"2015","unstructured":"Bai, L., Li, K., Pei, J., Jiang, S.: Main objects interaction activity recognition in real images. Neural Comput. Appl. 27, 335\u2013348 (2015). https:\/\/doi.org\/10.1007\/s00521-015-1846-7","journal-title":"Neural Comput. Appl."},{"doi-asserted-by":"publisher","unstructured":"Li J, Li D, Xiong C, Hoi S (2022) BLIP: bootstrapping language-image pre-training for unified vision-language understanding and generation. In: Proceedings of the 39th International Conference on Machine Learning. 162: 12888\u201312900. https:\/\/doi.org\/10.48550\/arXiv.2201.12086","key":"1610_CR13","DOI":"10.48550\/arXiv.2201.12086"},{"doi-asserted-by":"publisher","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A, Weissenborn D, Zhai X, Unterthiner T, Dehghani M, Minderer M,Heigold G, Gelly S, Uszkoreit J (2020) An image is worth 16x16 words: Transformers for image recognition at scale. In: arXiv preprint arXiv:2010.11929. https:\/\/doi.org\/10.48550\/arXiv.2010.11929","key":"1610_CR14","DOI":"10.48550\/arXiv.2010.11929"},{"doi-asserted-by":"publisher","unstructured":"Devlin J, Chang MW, Lee K, Toutanova K (2018) BERT: Pre-training of deep bidirectional transformers for language understanding. In: arXiv preprint arXiv:1810.04805. https:\/\/doi.org\/10.48550\/arXiv.1810.04805","key":"1610_CR15","DOI":"10.48550\/arXiv.1810.04805"},{"issue":"8","key":"1610_CR16","first-page":"9","volume":"1","author":"A Radford","year":"2019","unstructured":"Radford, A., Wu, J., Child, R., Luan, D., Amodei, D., Sutskever, I.: Language models are unsupervised multitask learners. OpenAI Blog 1(8), 9 (2019)","journal-title":"OpenAI Blog"},{"key":"1610_CR17","doi-asserted-by":"publisher","first-page":"853","DOI":"10.1613\/jair.3994","volume":"47","author":"M Hodosh","year":"2013","unstructured":"Hodosh, M., Young, P., Hockenmaier, J.: Framing image description as a ranking task: data, models and evaluation metrics. J Artif Intell Res 47, 853\u2013899 (2013). https:\/\/doi.org\/10.1613\/jair.3994","journal-title":"J Artif Intell Res"},{"doi-asserted-by":"publisher","unstructured":"Plummer BA, Wang L, Cervantes CM, Caicedo JC, Hockenmaier J, Lazebnik S (2015) Flickr30k entities:Collecting region-to-phrase correspondences for richer image-to-sentence models. In: Proceedings of the IEEE International Conference on Computer Vision, pp 2641\u20132649. https:\/\/doi.org\/10.48550\/arXiv.1505.04870","key":"1610_CR18","DOI":"10.48550\/arXiv.1505.04870"},{"doi-asserted-by":"publisher","unstructured":"Lin TY, Maire M, Belongie S, Hays J, Perona P, Ramanan D, Doll\u00e1r P, Zitnick CL (2014) Microsoft COCO: Common objects in context. In: Fleet D, PajdlaT, Schiele B, Tuytelaars T (eds) Computer Vision\u2014ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6\u201312, 2014, Proceedings, Part V, vol 8693, pp 740\u2013755. Springer International Publishing. https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48","key":"1610_CR19","DOI":"10.1007\/978-3-319-10602-1_48"},{"doi-asserted-by":"publisher","unstructured":"Wu J, Zheng H, Zhao B, Li Y, Yan B, Liang R, Wang W, Zhou S, Lin G, Fu Y, Wang Y (2017) AI Challenger: A large-scale dataset for going deeper in image understanding. In: arXiv preprint arXiv:1711.06475. https:\/\/doi.org\/10.48550\/arXiv.1711.06475","key":"1610_CR20","DOI":"10.48550\/arXiv.1711.06475"},{"doi-asserted-by":"publisher","unstructured":"Kirillov A, Mintun E, Ravi N, Mao H, Rolland C, Gustafson L, Whitehead S, Berg AC, Lo WY, Doll\u00e1r P (2023) Segment anything. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 4015\u20134026. https:\/\/doi.org\/10.48550\/arXiv.2304.02643","key":"1610_CR21","DOI":"10.48550\/arXiv.2304.02643"},{"key":"1610_CR22","doi-asserted-by":"publisher","first-page":"5998","DOI":"10.48550\/arXiv.1706.03762","volume":"30","author":"A Vaswani","year":"2017","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., JonesL, G.A.N., Kaiser, \u0141, Polosukhin, I.: Attention is all you need. Adv Neural Inf Process Syst. 30, 5998\u20136008 (2017). https:\/\/doi.org\/10.48550\/arXiv.1706.03762","journal-title":"Adv Neural Inf Process Syst."},{"doi-asserted-by":"publisher","unstructured":"Zhang H, Koh JY, Baldridge J, Lee H, Yang Y (2021) Cross-modal contrastive learning for text-to-image generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 833\u2013842. https:\/\/doi.org\/10.48550\/arXiv.2101.04702","key":"1610_CR23","DOI":"10.48550\/arXiv.2101.04702"},{"doi-asserted-by":"publisher","unstructured":"Papineni K, Roukos S, Ward T, Zhu WJ (2002) BLEU: A method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, pp 311\u2013318. https:\/\/doi.org\/10.3115\/1073083.1073135","key":"1610_CR24","DOI":"10.3115\/1073083.1073135"},{"doi-asserted-by":"publisher","unstructured":"Banerjee S, Lavie A (2005) METEOR: An automaticmetric for MT evaluation with improved correlation with human judgments. In: Proceedings of the ACL Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and\/or Summarization, pp 65\u201372. https:\/\/doi.org\/10.3115\/1626355.1626389","key":"1610_CR25","DOI":"10.3115\/1626355.1626389"},{"unstructured":"Lin CY (2004) ROUGE: A package for automatic evaluation of summaries. In: Text Summarization Branches Out, pp 74\u201381.","key":"1610_CR26"},{"doi-asserted-by":"publisher","unstructured":"Vedantam R, Lawrence Zitnick C, Parikh D (2015) CIDEr: Consensus-based image description evaluation.In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 4566\u20134575. https:\/\/doi.org\/10.1109\/CVPR.2015.7299087","key":"1610_CR27","DOI":"10.1109\/CVPR.2015.7299087"},{"doi-asserted-by":"publisher","unstructured":"Anderson P, Fernando B, Johnson M, Gould S (2016)SPICE: Semantic propositional image caption evaluation. In: European Conference on Computer Vision, vol 14, pp 382\u2013398. Springer International Publishing. https:\/\/doi.org\/10.1007\/978-3-319-46454-1_24","key":"1610_CR28","DOI":"10.1007\/978-3-319-46454-1_24"},{"key":"1610_CR29","doi-asserted-by":"publisher","first-page":"460","DOI":"10.1016\/j.procs.2022.10.064","volume":"208","author":"M Liang","year":"2022","unstructured":"Liang, M., Niu, T.: Research on text classification techniques based on improved TF-IDF algorithm and LSTM inputs. Proc Comput Sci 208, 460\u2013470 (2022). https:\/\/doi.org\/10.1016\/j.procs.2022.10.064","journal-title":"Proc Comput Sci"},{"doi-asserted-by":"publisher","unstructured":"Powers DM (2020) Evaluation: From precision, recalland F-measure to ROC, informedness, markedness and correlation. In: arXiv preprint arXiv:2010.16061. https:\/\/doi.org\/10.48550\/arXiv.2010.16061","key":"1610_CR30","DOI":"10.48550\/arXiv.2010.16061"},{"doi-asserted-by":"publisher","unstructured":"Mokady R, Hertz A, Bermano AH (2021) CLIPCap: CLIP Prefix for Image Captioning. In: arXiv preprint arXiv:2111.09734. https:\/\/doi.org\/10.48550\/arXiv.2111.09734","key":"1610_CR31","DOI":"10.48550\/arXiv.2111.09734"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01610-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-024-01610-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01610-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,28]],"date-time":"2025-02-28T10:58:29Z","timestamp":1740740309000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-024-01610-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,1,8]]},"references-count":31,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2025,2]]}},"alternative-id":["1610"],"URL":"https:\/\/doi.org\/10.1007\/s00530-024-01610-7","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"type":"print","value":"0942-4962"},{"type":"electronic","value":"1432-1882"}],"subject":[],"published":{"date-parts":[[2025,1,8]]},"assertion":[{"value":"2 September 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 December 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 January 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"53"}}