{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,6]],"date-time":"2025-11-06T06:00:35Z","timestamp":1762408835657,"version":"build-2065373602"},"reference-count":85,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"Ascending SNU Future Leader Fellowship through Seoul National University","award":["0524-20230015"],"award-info":[{"award-number":["0524-20230015"]}]},{"name":"Basic Science Research Program through the National Research Foundation of Korea (NRF) funded by the Ministry of Education","award":["RS-2023-00274280"],"award-info":[{"award-number":["RS-2023-00274280"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2025]]},"DOI":"10.1109\/access.2025.3618700","type":"journal-article","created":{"date-parts":[[2025,10,7]],"date-time":"2025-10-07T17:49:41Z","timestamp":1759859381000},"page":"176751-176769","source":"Crossref","is-referenced-by-count":0,"title":["Exploring Multimodal Perception in Large Language Models Through Perceptual Strength Ratings"],"prefix":"10.1109","volume":"13","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3217-7586","authenticated-orcid":false,"given":"Jonghyun","family":"Lee","sequence":"first","affiliation":[{"name":"English Studies Major, Division of Global Studies, College of Global Business, Korea University, Sejong Campus, Sejong, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-3674-3096","authenticated-orcid":false,"given":"Dojun","family":"Park","sequence":"additional","affiliation":[{"name":"Artificial Intelligence Institute, Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiwoo","family":"Lee","sequence":"additional","affiliation":[{"name":"Department of German Language and Literature, Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-0154-5311","authenticated-orcid":false,"given":"Hoekeon","family":"Choi","sequence":"additional","affiliation":[{"name":"Brain and Humanities Laboratory, Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-0013-2513","authenticated-orcid":false,"given":"Sung-Eun","family":"Lee","sequence":"additional","affiliation":[{"name":"Artificial Intelligence Institute, Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.463"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6239"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.559"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1017\/s0140525x99002149"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1146\/annurev.psych.59.103006.093639"},{"volume-title":"Louder Than Words: The New Science of How The Mind Makes Meaning","year":"2012","author":"Bergen","key":"ref6"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1016\/B978-0-08-046616-3.00016-5"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.3758\/bf03196313"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1017\/s0140525x9900182x"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1515\/langcog-2012-0001"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1016\/0167-2789(90)90087-6"},{"key":"ref12","first-page":"8469","article-title":"PaLM-E: An embodied multimodal language model","volume-title":"Proc. 40th Int. Conf. Mach. Learn.","author":"Driess"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52729.2023.01457"},{"key":"ref14","first-page":"72096","article-title":"Language is not all you need: Aligning perception with language models","volume-title":"pROC. Adv. Neural Inf. Process. Syst.","author":"Huang"},{"key":"ref15","first-page":"11928","article-title":"Multimodal language models show evidence of embodied simulation","volume-title":"Proc. 2024 Joint Int. Conf. Comput. Linguistics, Lang. Resour. Eval.","author":"Jones"},{"key":"ref16","article-title":"Scaling laws for neural language models","author":"Kaplan","year":"2020","journal-title":"arXiv:2001.08361"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.3758\/s13428-019-01316-z"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.3758\/s13428-024-02337-z"},{"key":"ref19","article-title":"Does conceptual representation require embodiment? Insights from large language models","author":"Xu","year":"2023","journal-title":"arXiv:2305.19103"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1016\/j.cortex.2019.10.014"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1007\/s00426-020-01374-5"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1523\/jneurosci.3579-08.2008"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1162\/jocn_a_00473"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1016\/j.jphysparis.2008.03.004"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1162\/0898929054021102"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1016\/j.neuropsychologia.2014.06.019"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1002\/hbm.20950"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.1018033108"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.3758\/bf03210980"},{"volume-title":"Hello GPT-4o","year":"2024","key":"ref30"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW67362.2025.00147"},{"key":"ref32","article-title":"VisualBERT: A simple and performant baseline for vision and language","author":"Li","year":"2019","journal-title":"arXiv:1908.03557"},{"key":"ref33","first-page":"12888","article-title":"BLIP: Bootstrapping language-image pre-training for unified vision-language understanding and generation","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Li"},{"key":"ref34","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","author":"Radford"},{"key":"ref35","first-page":"23716","article-title":"Flamingo: A visual language model for few-shot learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Alayrac"},{"key":"ref36","article-title":"Qwen2-VL: Enhancing vision-language model\u2019s perception of the world at any resolution","author":"Wang","year":"2024","journal-title":"arXiv:2409.12191"},{"key":"ref37","article-title":"Pixtral 12B","author":"Agrawal","year":"2024","journal-title":"arXiv:2410.07073"},{"key":"ref38","article-title":"The Llama 3 herd of models","author":"Grattafiori","year":"2024","journal-title":"arXiv:2407.21783"},{"key":"ref39","article-title":"DeepSeek-VL2: Mixture-of-experts vision-language models for advanced multimodal understanding","author":"Wu","year":"2024","journal-title":"arXiv:2412.10302"},{"key":"ref40","article-title":"NVLM: Open frontier-class multimodal LLMs","author":"Dai","year":"2024","journal-title":"arXiv:2409.11402"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICME52920.2022.9859720"},{"key":"ref42","article-title":"Do neural language representations learn physical commonsense?","author":"Forbes","year":"2019","journal-title":"arXiv:1908.02899"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/d19-6001"},{"article-title":"Mapping language models to grounded conceptual spaces","volume-title":"Proc. 10th Int. Conf. Learn. Represent.","author":"Patel","key":"ref44"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.360"},{"key":"ref46","first-page":"869","article-title":"Structured, flexible, and robust: Benchmarking and improving large language models towards more human-like behavior in out-of-distribution reasoning tasks","volume-title":"Proc. Annu. Meeting Cogn. Sci. Soc.","author":"Collins"},{"article-title":"Language models represent space and time","volume-title":"Proc. 12th Int. Conf. Learn. Represent.","author":"Gurnee","key":"ref47"},{"key":"ref48","article-title":"Contextualized sensorimotor norms: Multi-dimensional measures of sensorimotor strength for ambiguous English words, in context","author":"Trott","year":"2022","journal-title":"arXiv:2203.05648"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1007\/s10936-017-9548-1"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.3758\/s13428-017-0852-3"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.3758\/s13428-021-01656-9"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.3758\/s13428-019-01337-8"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.3758\/s13428-019-01254-w"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.3389\/fpsyg.2021.667271"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0211336"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.3389\/fpsyg.2023.1188909"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1111\/cogs.12549"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1111\/j.1551-6709.2010.01157.x"},{"volume-title":"Llama 3.2: Revolutionizing Edge AI and Vision With Open, Customizable Models","year":"2024","key":"ref59"},{"volume-title":"The Llama 4 Herd: The Beginning of a New Era of Natively Multimodal AI Innovation","year":"2025","key":"ref60"},{"key":"ref61","article-title":"Qwen2.5-VL technical report","volume-title":"arXiv:2502.13923","author":"Bai","year":"2025"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.3390\/app14177782"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.1159"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1016\/j.neuron.2013.02.008"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1098\/rstb.2017.0143"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1007\/s00426-020-01438-6"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1038\/s41592-019-0686-2"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0121945"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1037\/0021-9010.72.1.146"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.18637\/jss.v082.i13"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.18637\/jss.v067.i01"},{"article-title":"rspeer\/wordfreq: V3.0","year":"2022","author":"Speer","key":"ref72"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.3765\/sp.9.17"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.3758\/s13428-019-01243-z"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1002\/cpe.3745"},{"key":"ref76","first-page":"95266","article-title":"MMLU-Pro: A more robust and challenging multi-task language understanding benchmark","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"37","author":"Wang"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00913"},{"volume-title":"Llama 3.3 70B Vs Llama 3.2 90B: Text Mastery or Visual Power","year":"2025","author":"Novita","key":"ref78"},{"key":"ref79","article-title":"Jina-clip-v2: Multilingual multimodal embeddings for text and images","author":"Koukounas","year":"2024","journal-title":"arXiv:2412.08802"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1109\/ICME57554.2024.10687993"},{"key":"ref81","article-title":"A survey on efficient inference for large language models","author":"Zhou","year":"2024","journal-title":"arXiv:2404.14294"},{"key":"ref82","article-title":"Beyond Chinchilla-optimal: Accounting for inference in language model scaling laws","author":"Sardana","year":"2023","journal-title":"arXiv:2401.00448"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.1109\/tmech.2025.3574943"},{"key":"ref84","article-title":"Multi-agent embodied AI: Advances and future directions","author":"Feng","year":"2025","journal-title":"arXiv:2505.05108"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.1038\/s42256-025-01005-x"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/6287639\/10820123\/11195117.pdf?arnumber=11195117","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,6]],"date-time":"2025-11-06T05:49:21Z","timestamp":1762408161000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11195117\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"references-count":85,"URL":"https:\/\/doi.org\/10.1109\/access.2025.3618700","relation":{},"ISSN":["2169-3536"],"issn-type":[{"type":"electronic","value":"2169-3536"}],"subject":[],"published":{"date-parts":[[2025]]}}}