{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,6]],"date-time":"2026-06-06T13:24:36Z","timestamp":1780752276232,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":51,"publisher":"ACM","funder":[{"name":"Institute of Information & communications Technology Planning & Evaluation (IITP)","award":["RS-2025-02653113"],"award-info":[{"award-number":["RS-2025-02653113"]}]},{"name":"Institute of Information & Communications Technology Planning & Evaluation(IITP)","award":["IITP-2025-RS-2020-II201819"],"award-info":[{"award-number":["IITP-2025-RS-2020-II201819"]}]},{"name":"National Research Foundation of Korea(NRF)","award":["RS-2025-0051864"],"award-info":[{"award-number":["RS-2025-0051864"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,15]]},"DOI":"10.1145\/3757377.3763849","type":"proceedings-article","created":{"date-parts":[[2025,12,8]],"date-time":"2025-12-08T16:30:41Z","timestamp":1765211441000},"page":"1-12","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["How Does a Virtual Agent Decide Where to Look? Symbolic Cognitive Reasoning for Embodied Head Rotation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-4754-5568","authenticated-orcid":false,"given":"Juyeong","family":"Hwang","sequence":"first","affiliation":[{"name":"Korea University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7681-2617","authenticated-orcid":false,"given":"Seong-Eun","family":"Hong","sequence":"additional","affiliation":[{"name":"Korea University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-7097-7957","authenticated-orcid":false,"given":"JaeYoung","family":"Seon","sequence":"additional","affiliation":[{"name":"Kyung Hee University, Yongin, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5292-4342","authenticated-orcid":false,"given":"HyeongYeop","family":"Kang","sequence":"additional","affiliation":[{"name":"Korea University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,12,14]]},"reference":[{"key":"e_1_3_3_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/IC3D.2017.8251913"},{"key":"e_1_3_3_2_3_1","doi-asserted-by":"crossref","unstructured":"Sileye\u00a0O Ba and Jean-Marc Odobez. 2008. Recognizing visual focus of attention from head pose in natural meetings. IEEE Transactions on Systems Man and Cybernetics Part B (Cybernetics) 39 1 (2008) 16\u201333.","DOI":"10.1109\/TSMCB.2008.927274"},{"key":"e_1_3_3_2_4_1","doi-asserted-by":"crossref","unstructured":"Richard\u00a0W Bohannon. 1997. Comfortable and maximum walking speed of adults aged 20\u201479 years: reference values and determinants. Age and ageing 26 1 (1997) 15\u201319.","DOI":"10.1093\/ageing\/26.1.15"},{"key":"e_1_3_3_2_5_1","doi-asserted-by":"crossref","unstructured":"Ethan\u00a0S Bromberg-Martin and Ilya\u00a0E Monosov. 2020. Neural circuitry of information seeking. Current Opinion in Behavioral Sciences 35 (2020) 62\u201370.","DOI":"10.1016\/j.cobeha.2020.07.006"},{"key":"e_1_3_3_2_6_1","doi-asserted-by":"crossref","unstructured":"Panayiotis Charalambous Julien Pettre Vassilis Vassiliades Yiorgos Chrysanthou and Nuria Pelechano. 2023. GREIL-crowds: crowd simulation with deep reinforcement learning and examples. ACM Transactions on Graphics (TOG) 42 4 (2023) 1\u201315.","DOI":"10.1145\/3592459"},{"key":"e_1_3_3_2_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3561975.3562941"},{"key":"e_1_3_3_2_8_1","first-page":"894","volume-title":"International conference on machine learning","author":"Cuturi Marco","year":"2017","unstructured":"Marco Cuturi and Mathieu Blondel. 2017. Soft-dtw: a differentiable loss function for time-series. In International conference on machine learning. PMLR, 894\u2013903."},{"key":"e_1_3_3_2_9_1","doi-asserted-by":"publisher","DOI":"10.5555\/1089508.1089543"},{"key":"e_1_3_3_2_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3097895.3097898"},{"key":"e_1_3_3_2_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3083165.3083180"},{"key":"e_1_3_3_2_12_1","unstructured":"Rao Fu Jingyu Liu Xilun Chen Yixin Nie and Wenhan Xiong. 2024. Scene-llm: Extending language model for 3d visual understanding and reasoning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.11401 (2024)."},{"key":"e_1_3_3_2_13_1","doi-asserted-by":"crossref","unstructured":"Gerard\u00a0E Grossman R\u00a0John Leigh Larry\u00a0A Abel Douglas\u00a0J Lanska and SE Thurston. 1988. Frequency and velocity of rotational head perturbations during locomotion. Experimental brain research 70 (1988) 470\u2013476.","DOI":"10.1007\/BF00247595"},{"key":"e_1_3_3_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00228"},{"key":"e_1_3_3_2_15_1","doi-asserted-by":"crossref","unstructured":"David Harris Mark Wilson and Samuel Vine. 2020. Development and validation of a simulation workload measure: the simulation task load index (SIM-TLX). Virtual Reality 24 4 (2020) 557\u2013566.","DOI":"10.1007\/s10055-019-00422-9"},{"key":"e_1_3_3_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.573"},{"key":"e_1_3_3_2_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3641519.3657516"},{"key":"e_1_3_3_2_18_1","doi-asserted-by":"crossref","unstructured":"Robert\u00a0S Kennedy Norman\u00a0E Lane Kevin\u00a0S Berbaum and Michael\u00a0G Lilienthal. 1993. Simulator sickness questionnaire: An enhanced method for quantifying simulator sickness. The international journal of aviation psychology 3 3 (1993) 203\u2013220.","DOI":"10.1207\/s15327108ijap0303_3"},{"key":"e_1_3_3_2_19_1","unstructured":"Moo\u00a0Jin Kim Karl Pertsch Siddharth Karamcheti Ted Xiao Ashwin Balakrishna Suraj Nair Rafael Rafailov Ethan Foster Grace Lam Pannag Sanketi et\u00a0al. 2024. Openvla: An open-source vision-language-action model. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.09246 (2024)."},{"key":"e_1_3_3_2_20_1","doi-asserted-by":"publisher","DOI":"10.1111\/j.1467-8659.2007.01089.x"},{"key":"e_1_3_3_2_21_1","doi-asserted-by":"publisher","DOI":"10.1111\/j.1467-8659.2010.01808.x"},{"key":"e_1_3_3_2_22_1","doi-asserted-by":"crossref","unstructured":"Hugo Lerogeron Romain Picot-Clemente Alain Rakotomamonjy and Laurent Heutte. 2023. Approximating DTW with a convolutional neural network on EEG data. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2301.12873 (2023).","DOI":"10.1016\/j.patrec.2023.05.012"},{"key":"e_1_3_3_2_23_1","doi-asserted-by":"crossref","unstructured":"Huaizhou Li and Haiyan Hu. 2024. Head gesture recognition combining activity detection and dynamic time warping. Journal of Imaging 10 5 (2024) 123.","DOI":"10.3390\/jimaging10050123"},{"key":"e_1_3_3_2_24_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.440"},{"key":"e_1_3_3_2_25_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73404-5_12"},{"key":"e_1_3_3_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01495"},{"key":"e_1_3_3_2_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/2993369.2993378"},{"key":"e_1_3_3_2_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240508.3240669"},{"key":"e_1_3_3_2_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3528233.3530712"},{"key":"e_1_3_3_2_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/2980055.2980056"},{"key":"e_1_3_3_2_31_1","unstructured":"MFR Rondon L Sassatelli R Aparicio-Pardo and F Precioso. 2022. TRACK: A New Method From a Re-Examination of Deep Architectures for Head Motion Prediction in 360\u00b0 Videos. IEEE Transactions on Pattern Analysis and Machine Intelligence 44 9 (2022) 5681\u20135699."},{"key":"e_1_3_3_2_32_1","doi-asserted-by":"crossref","unstructured":"Hiroaki Sakoe and Seibi Chiba. 1978. Dynamic programming algorithm optimization for spoken word recognition. IEEE transactions on acoustics speech and signal processing 26 1 (1978) 43\u201349.","DOI":"10.1109\/TASSP.1978.1163055"},{"key":"e_1_3_3_2_33_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i22.34496"},{"key":"e_1_3_3_2_34_1","unstructured":"Noah Shinn Federico Cassano Ashwin Gopinath Karthik Narasimhan and Shunyu Yao. 2023. Reflexion: Language agents with verbal reinforcement learning. Advances in Neural Information Processing Systems 36 (2023) 8634\u20138652."},{"key":"e_1_3_3_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00280"},{"key":"e_1_3_3_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICAS.2010.36"},{"key":"e_1_3_3_2_37_1","doi-asserted-by":"crossref","unstructured":"Rainer Stiefelhagen Jie Yang and Alex Waibel. 2002. Modeling focus of attention for meeting indexing based on multiple cues. IEEE Transactions on Neural Networks 13 4 (2002) 928\u2013938.","DOI":"10.1109\/TNN.2002.1021893"},{"key":"e_1_3_3_2_38_1","doi-asserted-by":"crossref","unstructured":"Adam Switonski Henryk Josinski and Konrad Wojciechowski. 2019. Dynamic time warping in classification and selection of motion capture data. Multidimensional Systems and Signal Processing 30 (2019) 1437\u20131468.","DOI":"10.1007\/s11045-018-0611-3"},{"key":"e_1_3_3_2_39_1","unstructured":"Gemini Team Petko Georgiev Ving\u00a0Ian Lei Ryan Burnell Libin Bai Anmol Gulati Garrett Tanzer Damien Vincent Zhufeng Pan Shibo Wang et\u00a0al. 2024. Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.05530 (2024)."},{"key":"e_1_3_3_2_40_1","unstructured":"Frank Thomas. 1995. The illusion of life. (1995)."},{"key":"e_1_3_3_2_41_1","doi-asserted-by":"crossref","unstructured":"Zhou Wang Alan\u00a0C Bovik Hamid\u00a0R Sheikh and Eero\u00a0P Simoncelli. 2004. Image quality assessment: from error visibility to structural similarity. IEEE transactions on image processing 13 4 (2004) 600\u2013612.","DOI":"10.1109\/TIP.2003.819861"},{"key":"e_1_3_3_2_42_1","unstructured":"Jason Wei Xuezhi Wang Dale Schuurmans Maarten Bosma Fei Xia Ed Chi Quoc\u00a0V Le Denny Zhou et\u00a0al. 2022. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems 35 (2022) 24824\u201324837."},{"key":"e_1_3_3_2_43_1","doi-asserted-by":"crossref","unstructured":"Harry\u00a0J Witchel Carlos\u00a0P Santos James\u00a0K Ackah Carina\u00a0EI Westling and Nachiappan Chockalingam. 2016. Non-instrumental movement inhibition (NIMI) differentially suppresses head and thigh movements during screenic engagement: dependence on interaction. Frontiers in Psychology 7 (2016) 157.","DOI":"10.3389\/fpsyg.2016.00157"},{"key":"e_1_3_3_2_44_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20047-2_39"},{"key":"e_1_3_3_2_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00559"},{"key":"e_1_3_3_2_46_1","doi-asserted-by":"crossref","unstructured":"Li Yang Mai Xu Yichen Guo Xin Deng Fangyuan Gao and Zhenyu Guan. 2021. Hierarchical Bayesian LSTM for head trajectory prediction on omnidirectional images. IEEE Transactions on Pattern Analysis and Machine Intelligence 44 11 (2021) 7563\u20137580.","DOI":"10.1109\/TPAMI.2021.3117019"},{"key":"e_1_3_3_2_47_1","doi-asserted-by":"crossref","unstructured":"Yue Yang Yee\u00a0Mun Lee Ruth Madigan Albert Solernou and Natasha Merat. 2024. Interpreting pedestrians\u2019 head movements when encountering automated vehicles at a virtual crossroad. Transportation research part F: traffic psychology and behaviour 103 (2024) 340\u2013352.","DOI":"10.1016\/j.trf.2024.04.022"},{"key":"e_1_3_3_2_48_1","unstructured":"Zhengyuan Yang Linjie Li Jianfeng Wang Kevin Lin Ehsan Azarnasab Faisal Ahmed Zicheng Liu Ce Liu Michael Zeng and Lijuan Wang. 2023. Mm-react: Prompting chatgpt for multimodal reasoning and action. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.11381 (2023)."},{"key":"e_1_3_3_2_49_1","volume-title":"International Conference on Learning Representations (ICLR)","author":"Yao Shunyu","year":"2023","unstructured":"Shunyu Yao, Jeffrey Zhao, Dian Yu, Nan Du, Izhak Shafran, Karthik Narasimhan, and Yuan Cao. 2023. React: Synergizing reasoning and acting in language models. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_3_2_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS48716.2020.243571"},{"key":"e_1_3_3_2_51_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19830-4_22"},{"key":"e_1_3_3_2_52_1","doi-asserted-by":"crossref","unstructured":"Yucheng Zhu Guangtao Zhai Xiongkuo Min and Jiantao Zhou. 2020. Learning a deep agent to predict head movement in 360-degree images. ACM Transactions on Multimedia Computing Communications and Applications (TOMM) 16 4 (2020) 1\u201323.","DOI":"10.1145\/3410455"}],"event":{"name":"SA Conference Papers '25: SIGGRAPH Asia 2025 Conference Papers","location":"Hong Kong Hong Kong","acronym":"SA Conference Papers '25","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Proceedings of the SIGGRAPH Asia 2025 Conference Papers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3757377.3763849","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T03:28:46Z","timestamp":1765250926000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3757377.3763849"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,14]]},"references-count":51,"alternative-id":["10.1145\/3757377.3763849","10.1145\/3757377"],"URL":"https:\/\/doi.org\/10.1145\/3757377.3763849","relation":{},"subject":[],"published":{"date-parts":[[2025,12,14]]},"assertion":[{"value":"2025-12-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}