{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T08:18:02Z","timestamp":1783066682055,"version":"3.54.6"},"reference-count":41,"publisher":"Association for Computing Machinery (ACM)","issue":"4","license":[{"start":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T00:00:00Z","timestamp":1783036800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["2430673"],"award-info":[{"award-number":["2430673"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["2418236"],"award-info":[{"award-number":["2418236"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":["ACM Trans. Graph."],"published-print":{"date-parts":[[2026,7,3]]},"abstract":"<jats:p>We present a role-aware virtual agent navigational interaction that generates consistent, role-aligned movement behaviors. Our approach leverages Multimodal Large Language Models (MLLMs) to interpret multimodal inputs including scene information, user state, and high-level language role instruction, producing discrete navigation decisions and stylized planning path. Our approach enables virtual agents to behave consistently with narrative roles and respond to dynamic actions, such as playing a hide-and-seek taking into account the agent's role and the user's possible intention. Our approach demonstrates how MLLMs can go beyond language-based interaction to support embodied, spatial, and role-aware agent behaviors in immersive environments such as augmented reality.<\/jats:p>","DOI":"10.1145\/3811319","type":"journal-article","created":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T07:05:51Z","timestamp":1783062351000},"page":"1-16","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Role-Aware Virtual Agents for Navigational Interaction guided by a Multimodal Large Language Model"],"prefix":"10.1145","volume":"45","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4273-7118","authenticated-orcid":false,"given":"Minyoung","family":"Kim","sequence":"first","affiliation":[{"name":"George Mason University, Fairfax, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0074-3889","authenticated-orcid":false,"given":"Changyang","family":"Li","sequence":"additional","affiliation":[{"name":"Goertek Alpha Labs, Santa Clara, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9234-9960","authenticated-orcid":false,"given":"Cuong","family":"Nguyen","sequence":"additional","affiliation":[{"name":"Adobe Research, San Francisco, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2656-5654","authenticated-orcid":false,"given":"Lap-Fai","family":"Yu","sequence":"additional","affiliation":[{"name":"George Mason University, Fairfax, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,3]]},"reference":[{"key":"e_1_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2020.3038137"},{"key":"e_1_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3274247.3274511"},{"key":"e_1_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISMAR50242.2020.00020"},{"key":"e_1_2_2_4_1","volume-title":"Bojan Vujatovic, Bonnie Li, et al.","author":"Bolton Adrian","year":"2025","unstructured":"Adrian Bolton, Alexander Lerchner, Alexandra Cordell, Alexandre Moufarek, Andrew Bolt, Andrew Lampinen, Anna Mitenkova, Arne Olav Hallingstad, Bojan Vujatovic, Bonnie Li, et al. 2025. Sima 2: A generalist embodied agent for virtual worlds. arXiv preprint arXiv:2512.04797 (2025)."},{"key":"e_1_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.5555\/3463952.3463985"},{"key":"e_1_2_2_6_1","volume-title":"Tanishq Chawla, Zixin Zhuang, and Corey Clark.","author":"Buongiorno Steph","year":"2024","unstructured":"Steph Buongiorno, Lawrence Jake Klinkert, Tanishq Chawla, Zixin Zhuang, and Corey Clark. 2024. Pangea: Procedural artificial narrative using generative ai for turn-based video games. arXiv preprint arXiv:2404.19721 (2024)."},{"key":"e_1_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISMAR52148.2021.00045"},{"key":"e_1_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2025.3550231"},{"key":"e_1_2_2_9_1","volume-title":"Vi-LAD: Vision-Language Attention Distillation for Socially-Aware Robot Navigation in Dynamic Environments. arXiv preprint arXiv:2503.09820","author":"Elnoor Mohamed","year":"2025","unstructured":"Mohamed Elnoor, Kasun Weerakoon, Gershom Seneviratne, Jing Liang, Vignesh Rajagopal, and Dinesh Manocha. 2025. Vi-LAD: Vision-Language Attention Distillation for Socially-Aware Robot Navigation in Dynamic Environments. arXiv preprint arXiv:2503.09820 (2025)."},{"key":"e_1_2_2_10_1","doi-asserted-by":"crossref","unstructured":"Maxime Garcia R\u00e9mi Ronfard and Marie-Paule Cani. 2019. Spatial Motion Doodles: Sketching Animation in VR Using Hand Gestures and Laban Motion Analysis. In Motion Interaction and Games. 1\u201310.","DOI":"10.1145\/3359566.3360061"},{"key":"e_1_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA57147.2024.10611090"},{"key":"e_1_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3313831.3376719"},{"key":"e_1_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3517593"},{"key":"e_1_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3610977.3634966"},{"key":"e_1_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642917"},{"key":"e_1_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3641519.3657397"},{"key":"e_1_2_2_17_1","volume-title":"Autospatial: Visual-language reasoning for social robot navigation through efficient spatial reasoning learning. arXiv preprint arXiv:2503.07557","author":"Kong Yangzhe","year":"2025","unstructured":"Yangzhe Kong, Daeun Song, Jing Liang, Dinesh Manocha, Ziyu Yao, and Xuesu Xiao. 2025. Autospatial: Visual-language reasoning for social robot navigation through efficient spatial reasoning learning. arXiv preprint arXiv:2503.07557 (2025)."},{"key":"e_1_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3550454.3555429"},{"key":"e_1_2_2_19_1","volume-title":"X's Day: Personality-Driven Virtual Human Behavior Generation","author":"Li Haoyang","year":"2025","unstructured":"Haoyang Li, Zan Wang, Wei Liang, and Yizhuo Wang. 2025a. X's Day: Personality-Driven Virtual Human Behavior Generation. IEEE Transactions on Visualization and Computer Graphics (TVCG) (2025)."},{"key":"e_1_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3580978"},{"key":"e_1_2_2_21_1","volume-title":"LLM-enhanced Scene Graph Learning for Household Rearrangement. In SIGGRAPH Asia 2024 Conference Papers. 1\u201311","author":"Li Wenhao","year":"2024","unstructured":"Wenhao Li, Zhiyuan Yu, Qijin She, Zhinan Yu, Yuqing Lan, Chenyang Zhu, Ruizhen Hu, and Kai Xu. 2024. LLM-enhanced Scene Graph Learning for Household Rearrangement. In SIGGRAPH Asia 2024 Conference Papers. 1\u201311."},{"key":"e_1_2_2_22_1","doi-asserted-by":"crossref","unstructured":"Ziming Li Huadong Zhang Chao Peng and Roshan Peiris. 2025b. Exploring Large Language Model-Driven Agents for Environment-Aware Spatial Interactions and Conversations in Virtual Reality Role-Play Scenarios. In 2025 IEEE Conference Virtual Reality and 3D User Interfaces (VR). IEEE 1\u201311.","DOI":"10.1109\/VR59515.2025.00025"},{"key":"e_1_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445532"},{"key":"e_1_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.65109\/ZDXP6361"},{"key":"e_1_2_2_25_1","doi-asserted-by":"crossref","unstructured":"Meta Fundamental AI Research Diplomacy Team (FAIR) Anton Bakhtin Noam Brown Emily Dinan Gabriele Farina Colin Flaherty Daniel Fried Andrew Goff Jonathan Gray Hengyuan Hu et al. 2022. Human-level play in the game of Diplomacy by combining language models with strategic reasoning. Science 378 6624 (2022) 1067\u20131074.","DOI":"10.1126\/science.ade9097"},{"key":"e_1_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.52202\/075280-1090"},{"key":"e_1_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3610977.3634941"},{"key":"e_1_2_2_28_1","volume-title":"International Conference on Machine Learning. PMLR.","author":"Nottingham Kolby","year":"2023","unstructured":"Kolby Nottingham, Prithviraj Ammanabrolu, Alane Suhr, Yejin Choi, Hannaneh Hajishirzi, Sameer Singh, and Roy Fox. 2023. Do embodied agents dream of pixelated sheep: Embodied decision making using language guided world modelling. In International Conference on Machine Learning. PMLR."},{"key":"e_1_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2016.7759511"},{"key":"e_1_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ultrasmedbio.2006.02.315"},{"key":"e_1_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00280"},{"key":"e_1_2_2_32_1","volume-title":"Xuesu Xiao, and Dinesh Manocha.","author":"Song Daeun","year":"2024","unstructured":"Daeun Song, Jing Liang, Amirreza Payandeh, Amir Hossain Raj, Xuesu Xiao, and Dinesh Manocha. 2024. VLM-Social-Nav: Socially Aware Robot Navigation through Scoring using Vision-Language Models. IEEE Robotics and Automation Letters (2024)."},{"key":"e_1_2_2_33_1","volume-title":"TR-LLM: Integrating Trajectory Data for Scene-Aware LLM-Based Human Action Prediction. arXiv preprint arXiv:2410.03993","author":"Takeyama Kojiro","year":"2024","unstructured":"Kojiro Takeyama, Yimeng Liu, and Misha Sra. 2024. TR-LLM: Integrating Trajectory Data for Scene-Aware LLM-Based Human Action Prediction. arXiv preprint arXiv:2410.03993 (2024)."},{"key":"e_1_2_2_34_1","unstructured":"Unity Technologies. 2020. Unity Perception Package. https:\/\/github.com\/Unity-Technologies\/com.unity.perception."},{"key":"e_1_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300511"},{"key":"e_1_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3649921.3656987"},{"key":"e_1_2_2_37_1","volume-title":"Denny Zhou, et al.","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Fei Xia, Ed Chi, Quoc V Le, Denny Zhou, et al. 2022. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems 35 (2022), 24824\u201324837."},{"key":"e_1_2_2_38_1","volume-title":"Set-of-mark prompting unleashes extraordinary visual grounding in gpt-4v. arXiv preprint arXiv:2310.11441","author":"Yang Jianwei","year":"2023","unstructured":"Jianwei Yang, Hao Zhang, Feng Li, Xueyan Zou, Chunyuan Li, and Jianfeng Gao. 2023. Set-of-mark prompting unleashes extraordinary visual grounding in gpt-4v. arXiv preprint arXiv:2310.11441 (2023)."},{"key":"e_1_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3386569.3392404"},{"key":"e_1_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISMAR52148.2021.00039"},{"key":"e_1_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i16.29710"}],"container-title":["ACM Transactions on Graphics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/dl.acm.org\/doi\/abs\/10.1145\/3811319","content-type":"text\/html","content-version":"vor","intended-application":"syndication"}],"deposited":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T07:36:58Z","timestamp":1783064218000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3811319"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,3]]},"references-count":41,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2026,7,3]]}},"alternative-id":["10.1145\/3811319"],"URL":"https:\/\/doi.org\/10.1145\/3811319","relation":{},"ISSN":["0730-0301","1557-7368"],"issn-type":[{"value":"0730-0301","type":"print"},{"value":"1557-7368","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,3]]},"assertion":[{"value":"2026-01-22","order":0,"name":"received","label":"Received","group":{"name":"publication_history","label":"Publication History"}},{"value":"2026-03-27","order":2,"name":"accepted","label":"Accepted","group":{"name":"publication_history","label":"Publication History"}},{"value":"2026-07-03","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}