{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,9]],"date-time":"2026-06-09T11:04:03Z","timestamp":1781003043524,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":100,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,4,13]],"date-time":"2026-04-13T00:00:00Z","timestamp":1776038400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,13]]},"DOI":"10.1145\/3772318.3790946","type":"proceedings-article","created":{"date-parts":[[2026,4,13]],"date-time":"2026-04-13T04:12:21Z","timestamp":1776053541000},"page":"1-16","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["How Humans Naturally Refer to Targets: Understanding Multimodal Instruction Patterns in Human\u2013Robot Interaction"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3950-0905","authenticated-orcid":false,"given":"Lesong","family":"Jia","sequence":"first","affiliation":[{"name":"University of Pittsburgh, Pittsburgh, Pennsylvania, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-6507-8218","authenticated-orcid":false,"given":"Makayla","family":"Chang","sequence":"additional","affiliation":[{"name":"University of Pittsburgh, Pittsburgh, Pennsylvania, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-7153-504X","authenticated-orcid":false,"given":"Yu","family":"Liu","sequence":"additional","affiliation":[{"name":"University of Maryland, Baltimore County, Baltimore, Maryland, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4383-2451","authenticated-orcid":false,"given":"Na","family":"Du","sequence":"additional","affiliation":[{"name":"School of Computing and Information, University of Pittsburgh, Pittsburgh, Pennsylvania, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,13]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"publisher","DOI":"10.1109\/CCSSP49278.2020.9151809"},{"key":"e_1_3_3_2_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2016.7487510"},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"publisher","unstructured":"Arezoo Alizadeh and A\u00a0John Van\u00a0Opstal. 2022. Dynamic control of eye-head gaze shifts by a spiking neural network model of the superior colliculus. Frontiers in Computational Neuroscience 16 (2022) 1040646. 10.3389\/fncom.2022.1040646","DOI":"10.3389\/fncom.2022.1040646"},{"key":"e_1_3_3_2_5_2","doi-asserted-by":"publisher","unstructured":"Ferran Argelaguet and Carlos Andujar. 2009. Efficient 3D pointing selection in cluttered virtual environments. IEEE Computer Graphics and Applications 29 6 (2009) 34\u201343. 10.1109\/MCG.2009.117","DOI":"10.1109\/MCG.2009.117"},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"publisher","unstructured":"Ferran Argelaguet and Carlos Andujar. 2013. A survey of 3D object selection techniques for virtual environments. Computers & Graphics 37 3 (2013) 121\u2013136. 10.1016\/j.cag.2012.12.003","DOI":"10.1016\/j.cag.2012.12.003"},{"key":"e_1_3_3_2_7_2","doi-asserted-by":"publisher","unstructured":"Gozdem Arikan Peter Boddy and Kenny\u00a0R Coventry. 2025. The relative importance of language gaze and gesture in deictic reference. Journal of Experimental Psychology: Learning Memory and Cognition (2025). 10.1037\/xlm0001465","DOI":"10.1037\/xlm0001465"},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"publisher","unstructured":"Ameer\u00a0A Badr and Alia\u00a0K Abdul-Hassan. 2020. A review on voice-based interface for human-robot interaction. Iraqi Journal for Electrical and Electronic Engineering 16 2 (2020) 1\u201312. 10.37917\/ijeee.16.2.10","DOI":"10.37917\/ijeee.16.2.10"},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"publisher","unstructured":"Max Bain Jaesung Huh Tengda Han and Andrew Zisserman. 2023. Whisperx: Time-accurate speech transcription of long-form audio. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.00747 (2023). 10.48550\/arXiv.2303.00747","DOI":"10.48550\/arXiv.2303.00747"},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","unstructured":"Adrian Bangerter. 2004. Using pointing and describing to achieve joint focus of attention in dialogue. Psychological science 15 6 (2004) 415\u2013419. 10.1111\/j.0956-7976.2004.00694.x","DOI":"10.1111\/j.0956-7976.2004.00694.x"},{"key":"e_1_3_3_2_11_2","doi-asserted-by":"publisher","unstructured":"B. Birch Caiti\u00a0S. Griffiths and A. Morgan. 2021. Environmental effects on reliability and accuracy of MFCC based voice recognition for industrial human-robot-interaction. Proceedings of the Institution of Mechanical Engineers Part B: Journal of Engineering Manufacture 235 (2021) 1939 \u2013 1948. 10.1177\/09544054211014492","DOI":"10.1177\/09544054211014492"},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"publisher","DOI":"10.1177\/154193120404800401"},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-23232-9_32"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"publisher","DOI":"10.1109\/FiCloud.2019.00050"},{"key":"e_1_3_3_2_15_2","doi-asserted-by":"publisher","unstructured":"Junlong Chen Jens Grubert and P.\u00a0O. Kristensson. 2024. Large Language Model-assisted Speech and Pointing Benefits Multiple 3D Object Selection in Virtual Reality. ArXiv abs\/2410.21091 (2024). 10.48550\/arxiv.2410.21091","DOI":"10.48550\/arxiv.2410.21091"},{"key":"e_1_3_3_2_16_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00142"},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"publisher","unstructured":"Choongho Chung and Sung-Hee Lee. 2024. Continuous Prediction of Pointing Targets With Motion and Eye-Tracking in Virtual Reality. IEEE Access 12 (2024) 5933\u20135946. 10.1109\/ACCESS.2024.3350788","DOI":"10.1109\/ACCESS.2024.3350788"},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"publisher","unstructured":"Herbert\u00a0H Clark and Deanna Wilkes-Gibbs. 1986. Referring as a collaborative process. Cognition 22 1 (1986) 1\u201339. 10.1016\/0010-0277(86)90010-7","DOI":"10.1016\/0010-0277(86)90010-7"},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"publisher","DOI":"10.4324\/9780203771587"},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"publisher","DOI":"10.1145\/3489849.3489853"},{"key":"e_1_3_3_2_21_2","unstructured":"Shujie Deng. 2018. Multimodal interactions in virtual environments using eye tracking and gesture control.Ph.\u00a0D. Dissertation. Bournemouth University."},{"key":"e_1_3_3_2_22_2","doi-asserted-by":"publisher","unstructured":"Xiaoxi Du Jinchun Wu Xinyi Tang Xiaolei Lv Lesong Jia and Chengqi Xue. 2025. Predicting User Attention States from Multimodal Eye\u2014Hand Data in VR Selection Tasks. Electronics 14 10 (2025) 2052. 10.3390\/electronics14102052","DOI":"10.3390\/electronics14102052"},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"publisher","DOI":"10.23919\/WAC.2018.8430412"},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"publisher","unstructured":"Franz Faul Edgar Erdfelder Axel Buchner and Albert-Georg Lang. 2009. Statistical power analyses using G* Power 3.1: Tests for correlation and regression analyses. Behavior research methods 41 4 (2009) 1149\u20131160. 10.3758\/BRM.41.4.1149","DOI":"10.3758\/BRM.41.4.1149"},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"publisher","DOI":"10.1145\/3568444.3568454"},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"publisher","unstructured":"F. Freigang S. Klett and S. Kopp. 2017. Pragmatic Multimodality: Effects of Nonverbal Cues of Focus and Certainty in a Virtual Human. (2017) 142\u2013155. 10.1007\/978-3-319-67401-8_16","DOI":"10.1007\/978-3-319-67401-8_16"},{"key":"e_1_3_3_2_27_2","doi-asserted-by":"publisher","DOI":"10.1145\/3677386.3682080"},{"key":"e_1_3_3_2_28_2","unstructured":"Cindy Gallois Tania Ogay and Howard Giles. 2005. Communication accommodation theory. Theorizing about intercultural communication (2005) 121\u2013148."},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"publisher","unstructured":"Ryan Ghamandi Yahya Hmaiti Mykola Maslych Ravi\u00a0Kiran Kattoju and Joseph\u00a0J LaViola\u00a0Jr. 2025. Towards Deeper Understanding of Natural User Interactions in Virtual Reality Based Assembly Tasks. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2508.17124 (2025). 10.48550\/arXiv.2508.17124","DOI":"10.48550\/arXiv.2508.17124"},{"key":"e_1_3_3_2_30_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642491"},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"publisher","unstructured":"Eric Godden William Steedman and Matthew\u00a0KXJ Pan. 2025. Robotic Characterization of Markerless Hand-Tracking on Meta Quest Pro and Quest 3 Virtual Reality Headsets. IEEE Transactions on Visualization and Computer Graphics (2025). 10.1109\/TVCG.2025.3549182","DOI":"10.1109\/TVCG.2025.3549182"},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"publisher","unstructured":"Jose G\u00f3mez-Poveda and Elena Gaudioso. 2016. Evaluation of temporal stability of eye tracking algorithms using webcams. Expert Systems with Applications 64 (2016) 69\u201383. 10.1016\/j.eswa.2016.07.029","DOI":"10.1016\/j.eswa.2016.07.029"},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"publisher","DOI":"10.1163\/9789004368811_003"},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"publisher","unstructured":"Stephanie Gross Brigitte Krenn and Matthias Scheutz. 2016. Multi-modal referring expressions in human-human task descriptions and their implications for human-robot interaction. Interaction Studies 17 2 (2016) 180\u2013210. 10.1075\/is.17.2.02gro","DOI":"10.1075\/is.17.2.02gro"},{"key":"e_1_3_3_2_35_2","volume-title":"Proceedings of the Annual Meeting of the Cognitive Science Society","volume":"45","author":"Gu Yan","year":"2023","unstructured":"Yan Gu et\u00a0al. 2023. A recipient design in multimodal language on TV: A comparison of child-directed and adult-directed broadcasting. In Proceedings of the Annual Meeting of the Cognitive Science Society , Vol.\u00a045."},{"key":"e_1_3_3_2_36_2","doi-asserted-by":"publisher","unstructured":"Nishan Gunawardena Jeewani\u00a0Anupama Ginige and Bahman Javadi. 2022. Eye-tracking technologies in mobile devices Using edge computing: a systematic review. Comput. Surveys 55 8 (2022) 1\u201333. 10.1145\/3546938","DOI":"10.1145\/3546938"},{"key":"e_1_3_3_2_37_2","doi-asserted-by":"publisher","unstructured":"Joy\u00a0E Hanna and Susan\u00a0E Brennan. 2007. Speakers\u2019 eye gaze disambiguates referring expressions early during face-to-face conversation. Journal of memory and language 57 4 (2007) 596\u2013615. 10.1016\/j.jml.2007.01.008","DOI":"10.1016\/j.jml.2007.01.008"},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"publisher","unstructured":"Michael\u00a0W Harvey Lorna\u00a0C Timmerman and Oscar\u00a0G VazQuez. 2019. College and career readiness knowledge and effectiveness: Findings from an initial inquiry in Indiana. Journal of Educational and Psychological Consultation 29 3 (2019) 260\u2013282. 10.1080\/10474412.2018.1522260","DOI":"10.1080\/10474412.2018.1522260"},{"key":"e_1_3_3_2_39_2","doi-asserted-by":"publisher","unstructured":"Andr\u00e9 Helgert Carolin Strassmann and S. Eimler. 2024. Unlocking Potentials of Virtual Reality as a Research Tool in Human-Robot Interaction: A Wizard-of-Oz Approach. Companion of the 2024 ACM\/IEEE International Conference on Human-Robot Interaction (2024). 10.1145\/3610978.3640741","DOI":"10.1145\/3610978.3640741"},{"key":"e_1_3_3_2_40_2","doi-asserted-by":"publisher","unstructured":"Jinuk Heo Hyelim Choi Yongseok Lee Hyunsu Kim Harim Ji Hyunreal Park Youngseon Lee Cheongkee Jung Hai-Nguyen Nguyen and Dongjun Lee. 2024. Hand Tracking: Survey. International Journal of Control Automation and Systems 22 6 (2024) 1761\u20131778. 10.1007\/s12555-024-0298-1","DOI":"10.1007\/s12555-024-0298-1"},{"key":"e_1_3_3_2_41_2","doi-asserted-by":"publisher","unstructured":"Md\u00a0Tanz\u0131b Hosain Mehedi\u00a0Hasan Anik Sadman Rafi Rana Tabassum Khaleque Insia and Md\u00a0Mehrab S\u0131dd\u0131ky. 2023. Path to gain functional transparency in artificial intelligence with meaningful explainability. Journal of metaverse 3 2 (2023) 166\u2013180. 10.57019\/jmv.1306685","DOI":"10.57019\/jmv.1306685"},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"publisher","DOI":"10.1093\/acprof:oso\/9780199764150.001.0001"},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICTC49870.2020.9289160"},{"key":"e_1_3_3_2_44_2","doi-asserted-by":"publisher","DOI":"10.1109\/HRI61500.2025.10974057"},{"key":"e_1_3_3_2_45_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA57147.2024.10610543"},{"key":"e_1_3_3_2_46_2","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3502134"},{"key":"e_1_3_3_2_47_2","unstructured":"Clyde Kluckhohn. 1950. Human behavior and the principle of least effort."},{"key":"e_1_3_3_2_48_2","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2018.8594043"},{"key":"e_1_3_3_2_49_2","doi-asserted-by":"publisher","unstructured":"Dani\u00ebl Lakens. 2013. Calculating and reporting effect sizes to facilitate cumulative science: a practical primer for t-tests and ANOVAs. Frontiers in psychology 4 (2013) 863. 10.3389\/fpsyg.2013.00863","DOI":"10.3389\/fpsyg.2013.00863"},{"key":"e_1_3_3_2_50_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642230"},{"key":"e_1_3_3_2_51_2","doi-asserted-by":"publisher","unstructured":"Stephen\u00a0C Levinson and Judith Holler. 2014. The origin of human multi-modal communication. Philosophical Transactions of the Royal Society B: Biological Sciences 369 1651 (2014) 20130302. 10.1098\/rstb.2013.0302","DOI":"10.1098\/rstb.2013.0302"},{"key":"e_1_3_3_2_52_2","doi-asserted-by":"publisher","DOI":"10.1109\/HRI.2019.8673116"},{"key":"e_1_3_3_2_53_2","doi-asserted-by":"publisher","unstructured":"Yang Li Jin Huang Feng Tian Hongan Wang and G. Dai. 2019. Gesture interaction in virtual reality. Virtual Real. Intell. Hardw. 1 (2019) 84\u2013112. 10.3724\/sp.j.2096-5796.2018.0006","DOI":"10.3724\/sp.j.2096-5796.2018.0006"},{"key":"e_1_3_3_2_54_2","first-page":"21","volume-title":"Virtual and adaptive environments","author":"Loomis Jack\u00a0M","year":"2003","unstructured":"Jack\u00a0M Loomis and Joshua\u00a0M Knapp. 2003. Visual perception of egocentric distance in real and virtual environments. In Virtual and adaptive environments. CRC Press, 21\u201346."},{"key":"e_1_3_3_2_55_2","doi-asserted-by":"publisher","unstructured":"Andy L\u00fccking Thies Pfeiffer and Hannes Rieser. 2015. Pointing and reference reconsidered. Journal of Pragmatics 77 (2015) 56\u201379. 10.1016\/j.pragma.2014.12.013","DOI":"10.1016\/j.pragma.2014.12.013"},{"key":"e_1_3_3_2_56_2","doi-asserted-by":"publisher","unstructured":"Kristine Lund. 2007. The importance of gaze and gesture in interactive multimodal explanation. Language Resources and Evaluation 41 3 (2007) 289\u2013303. 10.1007\/s10579-007-9058-0","DOI":"10.1007\/s10579-007-9058-0"},{"key":"e_1_3_3_2_57_2","doi-asserted-by":"publisher","unstructured":"Mathias\u00a0N Lystb\u00e6k Peter Rosenberg Ken Pfeuffer Jens\u00a0Emil Gr\u00f8nb\u00e6k and Hans Gellersen. 2022. Gaze-hand alignment: Combining eye gaze and mid-air pointing for interacting with menus in augmented reality. Proceedings of the ACM on Human-Computer Interaction 6 ETRA (2022) 1\u201318. 10.1145\/3530886","DOI":"10.1145\/3530886"},{"key":"e_1_3_3_2_58_2","doi-asserted-by":"publisher","DOI":"10.1145\/3173574.3174227"},{"key":"e_1_3_3_2_59_2","unstructured":"John\u00a0H McDonald. 2014. Handbook of biological statistics. (2014)."},{"key":"e_1_3_3_2_60_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCE-Berlin.2017.8210646"},{"key":"e_1_3_3_2_61_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-77915-5_5"},{"key":"e_1_3_3_2_62_2","doi-asserted-by":"publisher","unstructured":"K. Murakami T. Hasegawa Kousuke Shigematsu Fumichika Sueyasu Yasunobu Nohara B. Ahn and R. Kurazume. 2010. Position tracking system of everyday objects in an everyday environment. 2010 IEEE\/RSJ International Conference on Intelligent Robots and Systems (2010) 3712\u20133718. 10.1109\/iros.2010.5649340","DOI":"10.1109\/iros.2010.5649340"},{"key":"e_1_3_3_2_63_2","doi-asserted-by":"publisher","unstructured":"Adithyavairavan Murali A. Mousavian Clemens Eppner Adam Fishman and D. Fox. 2023. CabiNet: Scaling Neural Collision Detection for Object Rearrangement with Procedural Scene Generation. 2023 IEEE International Conference on Robotics and Automation (ICRA) (2023) 1866\u20131874. 10.1109\/icra48891.2023.10161528","DOI":"10.1109\/icra48891.2023.10161528"},{"key":"e_1_3_3_2_64_2","doi-asserted-by":"publisher","unstructured":"Margherita Murgiano Yasamin Motamedi and Gabriella Vigliocco. 2021. Situating Language in the Real-World: The Role of Multimodal Iconicity and Indexicality. Journal of Cognition (Aug 2021). 10.5334\/joc.113","DOI":"10.5334\/joc.113"},{"key":"e_1_3_3_2_65_2","doi-asserted-by":"publisher","DOI":"10.4324\/9781315773155"},{"key":"e_1_3_3_2_66_2","volume-title":"Vision science: Photons to phenomenology","author":"Palmer Stephen\u00a0E","year":"1999","unstructured":"Stephen\u00a0E Palmer. 1999. Vision science: Photons to phenomenology. MIT press."},{"key":"e_1_3_3_2_67_2","volume-title":"Human dimension and interior space: A source book of design reference standards","author":"Panero Julius","year":"1979","unstructured":"Julius Panero and Martin Zelnik. 1979. Human dimension and interior space: A source book of design reference standards. Watson-Guptill."},{"key":"e_1_3_3_2_68_2","doi-asserted-by":"publisher","unstructured":"Zi\u00a0Haur Pang Yahui Fu Divesh Lala Mikey Elmers K. Inoue and Tatsuya Kawahara. 2025. Does the Appearance of Autonomous Conversational Robots Affect User Spoken Behaviors in Real-World Conference Interactions? Proceedings of the Extended Abstracts of the CHI Conference on Human Factors in Computing Systems (2025). 10.1145\/3706599.3720179","DOI":"10.1145\/3706599.3720179"},{"key":"e_1_3_3_2_69_2","doi-asserted-by":"publisher","DOI":"10.1145\/3131277.3132180"},{"key":"e_1_3_3_2_70_2","doi-asserted-by":"publisher","unstructured":"Katrin Plaumann Matthias Weing Christian Winkler Michael M\u00fcller and Enrico Rukzio. 2018. Towards accurate cursorless pointing: the effects of ocular dominance and handedness. Personal and Ubiquitous Computing 22 4 (2018) 633\u2013646. 10.1007\/s00779-017-1100-7","DOI":"10.1007\/s00779-017-1100-7"},{"key":"e_1_3_3_2_71_2","doi-asserted-by":"publisher","unstructured":"S. Priyanayana B. Jayasekara and Ruwan\u00a0Chandra Gopura. 2022. Adapting concept of human-human multimodal interaction in human-robot applications. Bolgoda Plains (2022). 10.31705\/bprm.v2(2).2022.4","DOI":"10.31705\/bprm.v2(2).2022.4"},{"key":"e_1_3_3_2_72_2","doi-asserted-by":"publisher","unstructured":"Kun Qian Zhuoyang Zhang Wei Song and Jianfeng Liao. 2023. Gvgnet: Gaze-directed visual grounding for learning under-specified object referring intention. IEEE Robotics and Automation Letters 8 9 (2023) 5990\u20135997. 10.1109\/LRA.2023.3301294","DOI":"10.1109\/LRA.2023.3301294"},{"key":"e_1_3_3_2_73_2","doi-asserted-by":"publisher","unstructured":"F. Roider. 2021. Natural Multimodal Interaction in the Car - Generating Design Support for Speech Gesture and Gaze Interaction while Driving. (2021). 10.20378\/irb-51826","DOI":"10.20378\/irb-51826"},{"key":"e_1_3_3_2_74_2","doi-asserted-by":"publisher","unstructured":"Ofir Sadka J. Giron D. Friedman Oren Zuckerman and H. Erel. 2020. Virtual-reality as a Simulation Tool for Non-humanoid Social Robots. Extended Abstracts of the 2020 CHI Conference on Human Factors in Computing Systems (2020). 10.1145\/3334480.3382893","DOI":"10.1145\/3334480.3382893"},{"key":"e_1_3_3_2_75_2","doi-asserted-by":"crossref","unstructured":"Mark\u00a0S Sanders and Ernest\u00a0James McCormick. 1998. Human factors in engineering and design. Industrial Robot: An International Journal 25 2 (1998) 153\u2013153.","DOI":"10.1108\/ir.1998.25.2.153.2"},{"key":"e_1_3_3_2_76_2","doi-asserted-by":"publisher","DOI":"10.1145\/3242671.3242675"},{"key":"e_1_3_3_2_77_2","doi-asserted-by":"publisher","unstructured":"Katie Seaborn Norihisa\u00a0P Miyake Peter Pennefather and Mihoko Otake-Matsuura. 2021. Voice in human\u2014agent interaction: A survey. ACM Computing Surveys (CSUR) 54 4 (2021) 1\u201343. 10.1145\/3386867","DOI":"10.1145\/3386867"},{"key":"e_1_3_3_2_78_2","doi-asserted-by":"publisher","DOI":"10.1145\/3706598.3713777"},{"key":"e_1_3_3_2_79_2","doi-asserted-by":"publisher","unstructured":"Rongkai Shi Yushi Wei Xueying Qin Pan Hui and Hai-Ning Liang. 2023. Exploring gaze-assisted and hand-based region selection in augmented reality. Proceedings of the ACM on Human-Computer Interaction 7 ETRA (2023) 1\u201319. 10.1145\/3591129","DOI":"10.1145\/3591129"},{"key":"e_1_3_3_2_80_2","doi-asserted-by":"publisher","DOI":"10.1145\/3706598.3714266"},{"key":"e_1_3_3_2_81_2","doi-asserted-by":"publisher","unstructured":"Snehesh Shrestha Yantian Zha Saketh Banagiri Ge Gao Yiannis Aloimonos and Cornelia Fermuller. 2024. Natsgd: A dataset with speech gestures and demonstrations for robot learning in natural human-robot interaction. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.02274 (2024). 10.48550\/arXiv.2403.02274","DOI":"10.48550\/arXiv.2403.02274"},{"key":"e_1_3_3_2_82_2","first-page":"71","volume-title":"ISCA Tutorial and Research Workshop on Error Handling in Spoken Dialogue Systems","author":"Skantze Gabriel","year":"2003","unstructured":"Gabriel Skantze. 2003. Exploring human error handling strategies: Implications for spoken dialogue systems. In ISCA Tutorial and Research Workshop on Error Handling in Spoken Dialogue Systems. Centre for Speech Technology KTH Stockholm, Sweden, 71\u201376."},{"key":"e_1_3_3_2_83_2","doi-asserted-by":"publisher","unstructured":"Rainer Stiefelhagen Hazim\u00a0Kemal Ekenel Christian Fugen Petra Gieselmann Hartwig Holzapfel Florian Kraft Kai Nickel Michael Voit and Alex Waibel. 2007. Enabling multimodal human\u2014robot interaction for the karlsruhe humanoid robot. IEEE Transactions on Robotics 23 5 (2007) 840\u2013851. 10.1109\/TRO.2007.907484","DOI":"10.1109\/TRO.2007.907484"},{"key":"e_1_3_3_2_84_2","volume-title":"Understanding pragmatics","author":"Verschueren Jef","year":"1999","unstructured":"Jef Verschueren. 1999. Understanding pragmatics."},{"key":"e_1_3_3_2_85_2","doi-asserted-by":"publisher","unstructured":"Valeria Villani Beatrice Capelli and Lorenzo Sabattini. 2018. Use of Virtual Reality for the Evaluation of Human-Robot Interaction Systems in Complex Scenarios. 2018 27th IEEE International Symposium on Robot and Human Interactive Communication (RO-MAN) (2018) 422\u2013427. 10.1109\/roman.2018.8525738","DOI":"10.1109\/roman.2018.8525738"},{"key":"e_1_3_3_2_86_2","unstructured":"Conor Wallace and Berat\u00a0A Erol. 2018. Voice Activation Control with Digital Assistant for Humanoid Robot Torso. (2018)."},{"key":"e_1_3_3_2_87_2","doi-asserted-by":"publisher","unstructured":"Lihui Wang Sichao Liu Hongyi Liu and Xiangyu Wang. 2020. Overview of Human-Robot Collaboration in Manufacturing. (2020) 15\u201358. 10.1007\/978-3-030-46212-3_2","DOI":"10.1007\/978-3-030-46212-3_2"},{"key":"e_1_3_3_2_88_2","doi-asserted-by":"publisher","unstructured":"Lihui Wang J. V\u00e1ncza Z. Kem\u00e9ny and X. Wang. 2021. Future Research Directions on Human\u2013Robot Collaboration. (2021) 439\u2013448. 10.1007\/978-3-030-69178-3_18","DOI":"10.1007\/978-3-030-69178-3_18"},{"key":"e_1_3_3_2_89_2","doi-asserted-by":"publisher","unstructured":"Lu Wang Di Zhang Fangkai Yang Pu Zhao Jianfeng Liu Yuefeng Zhan Hao Sun Qingwei Lin Weiwei Deng Dongmei Zhang Feng Sun and Qi Zhang. 2025. LettinGo: Explore User Profile Generation for Recommendation System. ArXiv abs\/2506.18309 (2025). 10.48550\/arxiv.2506.18309","DOI":"10.48550\/arxiv.2506.18309"},{"key":"e_1_3_3_2_90_2","doi-asserted-by":"publisher","unstructured":"Tian Wang Pai Zheng Shufei Li and Lihui Wang. 2024. Multimodal human\u2014robot interaction for human-centric smart manufacturing: a survey. Advanced Intelligent Systems 6 3 (2024) 2300359. 10.1002\/aisy.202300359","DOI":"10.1002\/aisy.202300359"},{"key":"e_1_3_3_2_91_2","doi-asserted-by":"publisher","unstructured":"Eleanor Watson Minh Nguyen Sarah Pan and Shujun Zhang. 2025. Choice Vectors: Streamlining Personal AI Alignment Through Binary Selection. Multimodal Technol. Interact. 9 (2025) 22. 10.3390\/mti9030022","DOI":"10.3390\/mti9030022"},{"key":"e_1_3_3_2_92_2","doi-asserted-by":"publisher","DOI":"10.1145\/3573381.3596467"},{"key":"e_1_3_3_2_93_2","doi-asserted-by":"publisher","unstructured":"Alex Wilson and Jessica Boehland. 2005. Small is beautiful US house size resource use and the environment. Journal of Industrial Ecology 9 1-2 (2005) 277\u2013287. 10.1162\/1088198054084680","DOI":"10.1162\/1088198054084680"},{"key":"e_1_3_3_2_94_2","doi-asserted-by":"publisher","unstructured":"Yoko Yamakata Tatsuya Kawahara Hiroshi\u00a0G Okuno and Michihiko Minoh. 2004. Belief network based disambiguation of object reference in spoken dialogue system. Transactions of the Japanese Society for Artificial Intelligence 19 1 (2004) 47\u201356. 10.1527\/tjsai.19.47","DOI":"10.1527\/tjsai.19.47"},{"key":"e_1_3_3_2_95_2","volume-title":"Proceedings of the Annual Meeting of the Cognitive Science Society","volume":"34","author":"Yasuda Tetsuya","year":"2012","unstructured":"Tetsuya Yasuda and Harumi Kobayashi. 2012. Roles of Adult\u2019s Gestures and Eye Gaze in Whole or Object Part Presenting. In Proceedings of the Annual Meeting of the Cognitive Science Society , Vol.\u00a034."},{"key":"e_1_3_3_2_96_2","doi-asserted-by":"crossref","unstructured":"James\u00a0W Youdas Tom\u00a0R Garrett Vera\u00a0J Suman Connie\u00a0L Bogard Horace\u00a0O Hallman James\u00a0R Carey et\u00a0al. 1992. Normal range of motion of the cervical spine: an initial goniometric study. Physical therapy 72 (1992) 770\u2013770.","DOI":"10.1093\/ptj\/72.11.770"},{"key":"e_1_3_3_2_97_2","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445343"},{"key":"e_1_3_3_2_98_2","first-page":"131\u20131\u2013131\u20138","volume-title":"Proceedings of the 2nd ICAUD International Conference in Architecture and Urban Design","author":"Yunitsyna Anna","year":"2014","unstructured":"Anna Yunitsyna. 2014. Universal Space in Dwelling \u2014 the Room for All Living Needs. In Proceedings of the 2nd ICAUD International Conference in Architecture and Urban Design. Epoka University, Tirana, Albania, 131\u20131\u2013131\u20138. Paper No. 131."},{"key":"e_1_3_3_2_99_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-22698-9_39"},{"key":"e_1_3_3_2_100_2","doi-asserted-by":"publisher","unstructured":"Qijie Zhao Xinming Yuan Dawei Tu and Jianxia Lu. 2015. Eye moving behaviors identification for gaze tracking interaction. Journal on Multimodal User Interfaces 9 2 (2015) 89\u2013104. 10.1007\/s12193-014-0171-2","DOI":"10.1007\/s12193-014-0171-2"},{"key":"e_1_3_3_2_101_2","doi-asserted-by":"publisher","unstructured":"Xiyuan Zhao Huijun Li Tianyuan Miao Xianyi Zhu Zhikai Wei Lifen Tan and Aiguo Song. 2024. Learning multimodal confidence for intention recognition in human-robot interaction. IEEE Robotics and Automation Letters (2024). 10.1109\/LRA.2024.3432352","DOI":"10.1109\/LRA.2024.3432352"}],"event":{"name":"CHI 2026: CHI Conference on Human Factors in Computing Systems","location":"Barcelona Spain","acronym":"CHI '26","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 2026 CHI Conference on Human Factors in Computing Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3772318.3790946","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,9]],"date-time":"2026-06-09T10:53:16Z","timestamp":1781002396000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3772318.3790946"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,13]]},"references-count":100,"alternative-id":["10.1145\/3772318.3790946","10.1145\/3772318"],"URL":"https:\/\/doi.org\/10.1145\/3772318.3790946","relation":{},"subject":[],"published":{"date-parts":[[2026,4,13]]},"assertion":[{"value":"2026-04-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}