{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:00:28Z","timestamp":1765339228119,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":53,"publisher":"ACM","funder":[{"name":"National Natural Science Foundation of China","award":["62372468"],"award-info":[{"award-number":["62372468"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755744","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:55:00Z","timestamp":1761375300000},"page":"5090-5099","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Reasoning Like Experts: Leveraging Multimodal Large Language Models for Drawing-based Psychoanalysis"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1908-3187","authenticated-orcid":false,"given":"Xueqi","family":"Ma","sequence":"first","affiliation":[{"name":"The University of Melbourne, Melbourne, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-6987-4138","authenticated-orcid":false,"given":"Yanbei","family":"Jiang","sequence":"additional","affiliation":[{"name":"The University of Melbourne, Melbourne, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0885-0643","authenticated-orcid":false,"given":"Sarah","family":"Erfani","sequence":"additional","affiliation":[{"name":"The University of Melbourne, Melbourne, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3769-3811","authenticated-orcid":false,"given":"James","family":"Bailey","sequence":"additional","affiliation":[{"name":"The University of Melbourne, Melbourne, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5388-9080","authenticated-orcid":false,"given":"Weifeng","family":"Liu","sequence":"additional","affiliation":[{"name":"China University of Petroleum (East China), Qingdao, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2247-3020","authenticated-orcid":false,"given":"Krista A.","family":"Ehinger","sequence":"additional","affiliation":[{"name":"The University of Melbourne, Melbourne, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1647-4628","authenticated-orcid":false,"given":"Jey Han","family":"Lau","sequence":"additional","affiliation":[{"name":"The University of Melbourne, Melbourne, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Jean-Baptiste Alayrac Jeff Donahue Pauline Luc Antoine Miech Iain Barr Yana Hasson Karel Lenc Arthur Mensch Katherine Millican Malcolm Reynolds et al. 2022. Flamingo: a visual language model for few-shot learning. Advances in neural information processing systems Vol. 35 (2022) 23716-23736."},{"key":"e_1_3_2_1_2_1","volume-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in neural information processing systems","author":"Baevski Alexei","year":"2020","unstructured":"Alexei Baevski, Yuhao Zhou, Abdelrahman Mohamed, and Michael Auli. 2020. wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in neural information processing systems, Vol. 33 (2020), 12449-12460."},{"key":"e_1_3_2_1_3_1","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Sibo Song Kai Dang Peng Wang Shijie Wang Jun Tang et al. 2025. Qwen2. 5-vl technical report. arXiv preprint arXiv:2502.13923 (2025)."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.3389\/frai.2025.1558696"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0165-0327(00)00358-X"},{"key":"e_1_3_2_1_6_1","volume-title":"Improving image captioning descriptiveness by ranking and llm-based fusion. arXiv preprint arXiv:2306.11593","author":"Bianco Simone","year":"2023","unstructured":"Simone Bianco, Luigi Celona, Marco Donzella, and Paolo Napoletano. 2023. Improving image captioning descriptiveness by ranking and llm-based fusion. arXiv preprint arXiv:2306.11593 (2023)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1002\/1097-4679(194804)4:2<151::AID-JCLP2270040203>3.0.CO;2-O"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10803-020-04771-2"},{"key":"e_1_3_2_1_9_1","volume-title":"Pali: A jointly-scaled multilingual language-image model. arXiv preprint arXiv:2209.06794","author":"Chen Xi","year":"2022","unstructured":"Xi Chen, Xiao Wang, Soravit Changpinyo, AJ Piergiovanni, Piotr Padlewski, Daniel Salz, Sebastian Goodman, Adam Grycner, Basil Mustafa, Lucas Beyer, et al., 2022. Pali: A jointly-scaled multilingual language-image model. arXiv preprint arXiv:2209.06794 (2022)."},{"key":"e_1_3_2_1_10_1","volume-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 24185-24198","author":"Chen Zhe","year":"2024","unstructured":"Zhe Chen, Jiannan Wu, Wenhai Wang, Weijie Su, Guo Chen, Sen Xing, Muyan Zhong, Qinglong Zhang, Xizhou Zhu, Lewei Lu, et al., 2024. Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 24185-24198."},{"key":"e_1_3_2_1_11_1","first-page":"4","article-title":"Draw-A-Person: Screening procedure for emotional disturbance","volume":"1","author":"Crusco Miranda","year":"2013","unstructured":"Miranda Crusco. 2013. Draw-A-Person: Screening procedure for emotional disturbance. The Centre for Longitudinal Studies Institute of Education University of London, Vol. 1 (2013), 4-5.","journal-title":"The Centre for Longitudinal Studies Institute of Education University of London"},{"key":"e_1_3_2_1_12_1","volume-title":"VLRM: Vision-Language Models act as Reward Models for Image Captioning. arXiv preprint arXiv:2404.01911","author":"Dzabraev Maksim","year":"2024","unstructured":"Maksim Dzabraev, Alexander Kunitsyn, and Andrei Ivaniuta. 2024. VLRM: Vision-Language Models act as Reward Models for Image Captioning. arXiv preprint arXiv:2404.01911 (2024)."},{"key":"e_1_3_2_1_13_1","volume-title":"The Formal Elements Art Therapy Scale and ''draw a person picking an apple from a tree.''. Handbook of art therapy","author":"Gantt Linda","year":"2003","unstructured":"Linda Gantt and Carmello Tabone. 2003. The Formal Elements Art Therapy Scale and ''draw a person picking an apple from a tree.''. Handbook of art therapy (2003), 420-427."},{"key":"e_1_3_2_1_14_1","volume-title":"Rada Mihalcea, and Soujanya Poria.","author":"Ghosal Deepanway","year":"2023","unstructured":"Deepanway Ghosal, Navonil Majumder, Roy Ka-Wei Lee, Rada Mihalcea, and Soujanya Poria. 2023. Language guided visual question answering: Elevate your multimodal language model using knowledge-enriched prompts. arXiv preprint arXiv:2310.20159 (2023)."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.3389\/fpsyt.2022.1041770"},{"key":"e_1_3_2_1_16_1","unstructured":"Emanuel F Hammer. 1958. The clinical application of projective drawings. (1958)."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02510"},{"key":"e_1_3_2_1_18_1","volume-title":"Cogvideo: Large-scale pretraining for text-to-video generation via transformers. arXiv preprint arXiv:2205.15868","author":"Hong Wenyi","year":"2022","unstructured":"Wenyi Hong, Ming Ding, Wendi Zheng, Xinghan Liu, and Jie Tang. 2022. Cogvideo: Large-scale pretraining for text-to-video generation via transformers. arXiv preprint arXiv:2205.15868 (2022)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i3.27999"},{"key":"e_1_3_2_1_20_1","volume-title":"Proceedings of the Thirty-Third International Joint Conference on Artificial Intelligence. 7663-7671","author":"Jiang Yanbei","year":"2024","unstructured":"Yanbei Jiang, Krista A Ehinger, and Jey Han Lau. 2024. KALE: an artwork image captioning system augmented with heterogeneous graph. In Proceedings of the Thirty-Third International Joint Conference on Artificial Intelligence. 7663-7671."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01300"},{"key":"e_1_3_2_1_22_1","volume-title":"Workshop on Responsibly Building the Next Generation of Multimodal Foundational Models.","author":"Lauren\u00e7on Hugo","year":"2024","unstructured":"Hugo Lauren\u00e7on, Andr\u00e9s Marafioti, Victor Sanh, and L\u00e9o Tronchon. 2024. Building and better understanding vision-language models: insights and future directions. In Workshop on Responsibly Building the Next Generation of Multimodal Foundational Models."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.datak.2023.102266"},{"key":"e_1_3_2_1_24_1","volume-title":"International conference on machine learning. PMLR","author":"Li Junnan","year":"2023","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven Hoi. 2023. Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In International conference on machine learning. PMLR, 19730-19742."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.actpsy.2022.103734"},{"key":"e_1_3_2_1_26_1","volume-title":"European Conference on Computer Vision. Springer, 38-55","author":"Liu Shilong","year":"2024","unstructured":"Shilong Liu, Zhaoyang Zeng, Tianhe Ren, Feng Li, Hao Zhang, Jie Yang, Qing Jiang, Chunyuan Li, Jianwei Yang, Hang Su, et al., 2024b. Grounding dino: Marrying dino with grounded pre-training for open-set object detection. In European Conference on Computer Vision. Springer, 38-55."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671552"},{"key":"e_1_3_2_1_28_1","unstructured":"Haoyu Lu Wen Liu Bo Zhang Bingxuan Wang Kai Dong Bo Liu Jingxiang Sun Tongzheng Ren Zhuoshu Li Hao Yang et al. 2024. Deepseek-vl: towards real-world vision-language understanding. arXiv preprint arXiv:2403.05525 (2024)."},{"key":"e_1_3_2_1_29_1","volume-title":"Macaw-llm: Multi-modal language modeling with image, audio, video, and text integration. arXiv preprint arXiv:2306.09093","author":"Lyu Chenyang","year":"2023","unstructured":"Chenyang Lyu, Minghao Wu, Longyue Wang, Xinting Huang, Bingshuai Liu, Zefeng Du, Shuming Shi, and Zhaopeng Tu. 2023. Macaw-llm: Multi-modal language modeling with image, audio, video, and text integration. arXiv preprint arXiv:2306.09093 (2023)."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/1873951.1873965"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"crossref","unstructured":"Manfredo Massironi et al. 2001. The psychology of graphic images: Seeing drawing communicating. Psychology Press.","DOI":"10.4324\/9781410601896"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.02058"},{"key":"e_1_3_2_1_33_1","volume-title":"https:\/\/openai.com\/index\/hello-gpt-4o\/","author":"Blog AI.","year":"2024","unstructured":"OpenAI. 2024. Hello GPT-4o. OpenAI Blog (2024). https:\/\/openai.com\/index\/hello-gpt-4o\/"},{"volume-title":"Using drawings in assessment and therapy: A guide for mental health professionals","author":"Oster Gerald D","key":"e_1_3_2_1_34_1","unstructured":"Gerald D Oster and Patricia Gould Crone. 2004. Using drawings in assessment and therapy: A guide for mental health professionals. Routledge."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICTAI56018.2022.00171"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298687"},{"key":"e_1_3_2_1_37_1","volume-title":"The clinical and projective use of the Bender-Gestalt Test. (No Title)","author":"Perticone Eugene X","year":"1998","unstructured":"Eugene X Perticone and John B Ruskowski. 1998. The clinical and projective use of the Bender-Gestalt Test. (No Title) (1998)."},{"key":"e_1_3_2_1_38_1","volume-title":"International conference on machine learning. PmLR, 8748-8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al., 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PmLR, 8748-8763."},{"key":"e_1_3_2_1_39_1","volume-title":"International conference on machine learning. PMLR, 28492-28518","author":"Radford Alec","year":"2023","unstructured":"Alec Radford, Jong Wook Kim, Tao Xu, Greg Brockman, Christine McLeavey, and Ilya Sutskever. 2023. Robust speech recognition via large-scale weak supervision. In International conference on machine learning. PMLR, 28492-28518."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICECCME57830.2023.10252218"},{"key":"e_1_3_2_1_42_1","volume-title":"Deepseekmath: Pushing the limits of mathematical reasoning in open language models. arXiv preprint arXiv:2402.03300","author":"Shao Zhihong","year":"2024","unstructured":"Zhihong Shao, Peiyi Wang, Qihao Zhu, Runxin Xu, Junxiao Song, Xiao Bi, Haowei Zhang, Mingchuan Zhang, YK Li, Y Wu, et al., 2024. Deepseekmath: Pushing the limits of mathematical reasoning in open language models. arXiv preprint arXiv:2402.03300 (2024)."},{"key":"e_1_3_2_1_43_1","volume-title":"Few-shot vqa with frozen llms: A tale of two approaches. arXiv preprint arXiv:2403.11317","author":"Sterner Igor","year":"2024","unstructured":"Igor Sterner, Weizhe Lin, Jinghong Chen, and Bill Byrne. 2024. Few-shot vqa with frozen llms: A tale of two approaches. arXiv preprint arXiv:2403.11317 (2024)."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.3390\/fi16070247"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02408"},{"key":"e_1_3_2_1_46_1","volume-title":"Australasian Joint Conference on Artificial Intelligence. Springer, 221-233","author":"Xie Yaowu","year":"2023","unstructured":"Yaowu Xie, Ting Pan, Baodi Liu, Honglong Chen, and Weifeng Liu. 2023. Interpretable Drawing Psychoanalysis via House-Tree-Person Test. In Australasian Joint Conference on Artificial Intelligence. Springer, 221-233."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1007\/s00530-024-01610-7"},{"key":"e_1_3_2_1_48_1","unstructured":"An Yang Baosong Yang Beichen Zhang Binyuan Hui Bo Zheng Bowen Yu Chengyuan Li Dayiheng Liu Fei Huang Haoran Wei et al. 2024a. Qwen2. 5 technical report. arXiv preprint arXiv:2412.15115 (2024)."},{"key":"e_1_3_2_1_49_1","volume-title":"Emollm: Multimodal emotional understanding meets large language models. arXiv preprint arXiv:2406.16442","author":"Yang Qu","year":"2024","unstructured":"Qu Yang, Mang Ye, and Bo Du. 2024b. Emollm: Multimodal emotional understanding meets large language models. arXiv preprint arXiv:2406.16442 (2024)."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01337"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1080\/10255842.2023.2231113"},{"key":"e_1_3_2_1_52_1","volume-title":"Mlvu: A comprehensive benchmark for multi-task long video understanding. arXiv preprint arXiv:2406.04264","author":"Zhou Junjie","year":"2024","unstructured":"Junjie Zhou, Yan Shu, Bo Zhao, Boya Wu, Shitao Xiao, Xi Yang, Yongping Xiong, Bo Zhang, Tiejun Huang, and Zheng Liu. 2024. Mlvu: A comprehensive benchmark for multi-task long video understanding. arXiv preprint arXiv:2406.04264 (2024)."},{"key":"e_1_3_2_1_53_1","volume-title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models. arXiv preprint arXiv:2304.10592","author":"Zhu Deyao","year":"2023","unstructured":"Deyao Zhu, Jun Chen, Xiaoqian Shen, Xiang Li, and Mohamed Elhoseiny. 2023. Minigpt-4: Enhancing vision-language understanding with advanced large language models. arXiv preprint arXiv:2304.10592 (2023)."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755744","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T03:58:37Z","timestamp":1765339117000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755744"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":53,"alternative-id":["10.1145\/3746027.3755744","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755744","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}