{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,3]],"date-time":"2026-04-03T15:37:05Z","timestamp":1775230625442,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":71,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,9]],"date-time":"2023-10-09T00:00:00Z","timestamp":1696809600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nd\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,9]]},"DOI":"10.1145\/3610661.3616548","type":"proceedings-article","created":{"date-parts":[[2023,10,9]],"date-time":"2023-10-09T16:51:22Z","timestamp":1696870282000},"page":"164-173","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["A Methodology for Evaluating Multimodal Referring Expression Generation for Embodied Virtual Agents"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9253-416X","authenticated-orcid":false,"given":"Nada","family":"Alalyani","sequence":"first","affiliation":[{"name":"Situated Grounding and Natural Language (SIGNAL) Lab, Department of Computer Science, Colorado State University, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7878-7227","authenticated-orcid":false,"given":"Nikhil","family":"Krishnaswamy","sequence":"additional","affiliation":[{"name":"Situated Grounding and Natural Language (SIGNAL) Lab, Department of Computer Science, Colorado State University, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,10,9]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Julia Albath Jennifer\u00a0L Leopold Chaman\u00a0L Sabharwal and Anne\u00a0M Maglia. 2010. RCC-3D: Qualitative Spatial Reasoning in 3D.. In CAINE. 74\u201379."},{"key":"e_1_3_2_1_2_1","volume-title":"Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization. 65\u201372","author":"Banerjee Satanjeev","year":"2005","unstructured":"Satanjeev Banerjee and Alon Lavie. 2005. METEOR: An automatic metric for MT evaluation with improved correlation with human judgments. In Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization. 65\u201372."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.3115\/1557690.1557746"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.463"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-36272-9_69"},{"key":"e_1_3_2_1_6_1","volume-title":"AI and the limits of language. Noema Magazine","author":"Browning Jacob","year":"2022","unstructured":"Jacob Browning and Yann LeCun. 2022. AI and the limits of language. Noema Magazine (2022)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01282"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01010"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0022-5371(83)90189-5"},{"key":"e_1_3_2_1_10_1","volume-title":"Computational interpretations of the Gricean maxims in the generation of referring expressions. Cognitive science 19, 2","author":"Dale Robert","year":"1995","unstructured":"Robert Dale and Ehud Reiter. 1995. Computational interpretations of the Gricean maxims in the generation of referring expressions. Cognitive science 19, 2 (1995), 233\u2013263."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.475"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-1423"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.2307\/1932409"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS40897.2019.8968510"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/2696454.2696467"},{"key":"e_1_3_2_1_16_1","volume-title":"Statistical methods for research workers.Statistical methods for research workers","author":"Ronald\u00a0Aylmer","year":"1936","unstructured":"Ronald\u00a0Aylmer Fisher 1936. Statistical methods for research workers.Statistical methods for research workers.6th Ed (1936).","edition":"6"},{"key":"e_1_3_2_1_17_1","volume-title":"The TUNA-REG Challenge 2009: Overview and evaluation results","author":"Gatt Albert","unstructured":"Albert Gatt, Anja Belz, and Eric Kow. 2009. The TUNA-REG Challenge 2009: Overview and evaluation results. Association for Computational Linguistics."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10849-007-9047-0"},{"key":"e_1_3_2_1_19_1","volume-title":"The role of gesture in communication and thinking. Trends in cognitive sciences 3, 11","author":"Goldin-Meadow Susan","year":"1999","unstructured":"Susan Goldin-Meadow. 1999. The role of gesture in communication and thinking. Trends in cognitive sciences 3, 11 (1999), 419\u2013429."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.5555\/1622467.1622480"},{"key":"e_1_3_2_1_21_1","volume-title":"Speech acts","author":"Grice P","unstructured":"Herbert\u00a0P Grice. 1975. Logic and conversation. In Speech acts. Brill, 41\u201358."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1086"},{"key":"e_1_3_2_1_23_1","volume-title":"Assessing open-ended human-computer collaboration systems: applying a hallmarks approach. Frontiers in artificial intelligence 4","author":"Kozierok Robyn","year":"2021","unstructured":"Robyn Kozierok, John Aberdeen, Cheryl Clark, Christopher Garay, Bradley Goodman, Tonia Korves, Lynette Hirschman, Patricia\u00a0L McDermott, and Matthew\u00a0W Peterson. 2021. Assessing open-ended human-computer collaboration systems: applying a hallmarks approach. Frontiers in artificial intelligence 4 (2021), 670009."},{"key":"e_1_3_2_1_24_1","volume-title":"Proceedings of the ENLG, Vol.\u00a03. 47\u201354","author":"Krahmer Emiel","year":"2003","unstructured":"Emiel Krahmer and Ielka van\u00a0der Sluis. 2003. A new model for generating multimodal referring expressions. In Proceedings of the ENLG, Vol.\u00a03. 47\u201354."},{"key":"e_1_3_2_1_25_1","volume-title":"AAMAS\u201902 Workshop Embodied conversational agents-let\u2019s specify and evaluate them!","author":"Kranstedt Alfred","year":"2002","unstructured":"Alfred Kranstedt, Stefan Kopp, and Ipke Wachsmuth. 2002. Murml: A multimodal utterance representation markup language for conversational agents. In AAMAS\u201902 Workshop Embodied conversational agents-let\u2019s specify and evaluate them!"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1007\/11678816_34"},{"key":"e_1_3_2_1_27_1","volume-title":"Proceedings of the First Workshop on Bridging Human\u2013Computer Interaction and Natural Language Processing. Association for Computational Linguistics, Online, 41\u201346","author":"Krishnaswamy Nikhil","year":"2021","unstructured":"Nikhil Krishnaswamy and Nada Alalyani. 2021. Embodied Multimodal Agents to Bridge the Understanding Gap. In Proceedings of the First Workshop on Bridging Human\u2013Computer Interaction and Natural Language Processing. Association for Computational Linguistics, Online, 41\u201346. https:\/\/aclanthology.org\/2021.hcinlp-1.7"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i09.7096"},{"key":"e_1_3_2_1_29_1","volume-title":"Proceedings of the Thirteenth Language Resources and Evaluation Conference. 1529\u20131541","author":"Krishnaswamy Nikhil","year":"2022","unstructured":"Nikhil Krishnaswamy, William Pickard, Brittany Cates, Nathaniel Blanchard, and James Pustejovsky. 2022. The VoxWorld platform for multimodal embodied agents. In Proceedings of the Thirteenth Language Resources and Evaluation Conference. 1529\u20131541."},{"key":"e_1_3_2_1_30_1","volume-title":"Proceedings of COLING 2016, the 26th International Conference on Computational Linguistics: Technical Papers. ACL.","author":"Krishnaswamy Nikhil","year":"2016","unstructured":"Nikhil Krishnaswamy and James Pustejovsky. 2016. VoxSim: A Visual Platform for Modeling Motion Language. In Proceedings of COLING 2016, the 26th International Conference on Computational Linguistics: Technical Papers. ACL."},{"key":"e_1_3_2_1_31_1","volume-title":"Proceedings of the Eleventh International Conference on Language Resources and Evaluation (LREC","author":"Krishnaswamy Nikhil","year":"2018","unstructured":"Nikhil Krishnaswamy and James Pustejovsky. 2018. An evaluation framework for multimodal interaction. In Proceedings of the Eleventh International Conference on Language Resources and Evaluation (LREC 2018)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W19-0507"},{"key":"e_1_3_2_1_33_1","volume-title":"The Role of Embodiment and Simulation in Evaluating HCI: Experiments and Evaluation. In International Conference on Human-Computer Interaction. 220\u2013232","author":"Krishnaswamy Nikhil","year":"2021","unstructured":"Nikhil Krishnaswamy and James Pustejovsky. 2021. The Role of Embodiment and Simulation in Evaluating HCI: Experiments and Evaluation. In International Conference on Human-Computer Interaction. 220\u2013232."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.3389\/frai.2022.774752"},{"key":"e_1_3_2_1_35_1","volume-title":"2017 AAAI Fall Symposium Series.","author":"Kunze Lars","year":"2017","unstructured":"Lars Kunze, Tom Williams, Nick Hawes, and Matthias Scheutz. 2017. Spatial referring expression generation for hri: Algorithms and evaluation framework. In 2017 AAAI Fall Symposium Series."},{"key":"e_1_3_2_1_36_1","volume-title":"Workshop on Interoperable Semantic Annotation (ISA-19)","author":"Lee Kiyong","year":"2023","unstructured":"Kiyong Lee, Nikhil Krishnaswamy, and James Pustejovsky. 2023. An Abstract Specification of VoxML as an Annotation Language. In Workshop on Interoperable Semantic Annotation (ISA-19). 66."},{"key":"e_1_3_2_1_37_1","volume-title":"Soviet physics doklady, Vol.\u00a010","author":"I Levenshtein","unstructured":"Vladimir\u00a0I Levenshtein 1966. Binary codes capable of correcting deletions, insertions, and reversals. In Soviet physics doklady, Vol.\u00a010. Soviet Union, 707\u2013710."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2022.3141150"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.3115\/1073445.1073465"},{"key":"e_1_3_2_1_40_1","volume-title":"Conference on Robot Learning. PMLR, 76\u201385","author":"Magassouba Aly","year":"2020","unstructured":"Aly Magassouba, Komei Sugiura, and Hisashi Kawai. 2020. Multimodal attention branch network for perspective-free sentence generation. In Conference on Robot Learning. PMLR, 76\u201385."},{"key":"e_1_3_2_1_41_1","volume-title":"Dissociating language and thought in large language models: a cognitive perspective. arXiv preprint arXiv:2301.06627","author":"Mahowald Kyle","year":"2023","unstructured":"Kyle Mahowald, Anna\u00a0A Ivanova, Idan\u00a0A Blank, Nancy Kanwisher, Joshua\u00a0B Tenenbaum, and Evelina Fedorenko. 2023. Dissociating language and thought in large language models: a cognitive perspective. arXiv preprint arXiv:2301.06627 (2023)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/P14-5010"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.9"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/HCC46620.2019.00015"},{"key":"e_1_3_2_1_45_1","volume-title":"So you think gestures are nonverbal?Psychological review 92, 3","author":"McNeill David","year":"1985","unstructured":"David McNeill. 1985. So you think gestures are nonverbal?Psychological review 92, 3 (1985), 350."},{"key":"e_1_3_2_1_46_1","volume-title":"Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP). In Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP).","author":"Moschitti Alessandro","year":"2014","unstructured":"Alessandro Moschitti, Bo Pang, and Walter Daelemans. 2014. Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP). In Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP)."},{"key":"e_1_3_2_1_47_1","volume-title":"Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311\u2013318","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. Bleu: a method for automatic evaluation of machine translation. In Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311\u2013318."},{"key":"e_1_3_2_1_48_1","unstructured":"Rebecca Passonneau. 2006. Measuring agreement on set-valued items (MASI) for semantic and pragmatic annotation. (2006)."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.303"},{"key":"e_1_3_2_1_50_1","volume-title":"Proceedings of LREC","author":"Pustejovsky James","year":"2016","unstructured":"James Pustejovsky and Nikhil Krishnaswamy. 2016. VoxML: A Visualization Modeling Language. Proceedings of LREC (2016)."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3383652.3423910"},{"key":"e_1_3_2_1_52_1","first-page":"3","article-title":"Embodied human computer interaction","volume":"35","author":"Pustejovsky James","year":"2021","unstructured":"James Pustejovsky and Nikhil Krishnaswamy. 2021. Embodied human computer interaction. KI-K\u00fcnstliche Intelligenz 35, 3-4 (2021), 307\u2013327.","journal-title":"KI-K\u00fcnstliche Intelligenz"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-05311-5_9"},{"key":"e_1_3_2_1_54_1","volume-title":"AAAI Spring Symposium: Interactive Multisensory Object Perception for Embodied Agents.","author":"Pustejovsky James","year":"2017","unstructured":"James Pustejovsky, Nikhil Krishnaswamy, and Tuan Do. 2017. Object Embodiment in a Multimodal Simulation. In AAAI Spring Symposium: Interactive Multisensory Object Perception for Embodied Agents."},{"key":"e_1_3_2_1_55_1","volume-title":"Proceedings of the IWCS workshop on Foundations of Situated and Multimodal Communication.","author":"Pustejovsky James","year":"2017","unstructured":"James Pustejovsky, Nikhil Krishnaswamy, Bruce Draper, Pradyumna Narayana, and Rahul Bangar. 2017. Creating common ground through multimodal simulations. In Proceedings of the IWCS workshop on Foundations of Situated and Multimodal Communication."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01000"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/568513.568514"},{"key":"e_1_3_2_1_58_1","volume-title":"Language models are unsupervised multitask learners. OpenAI blog 1, 8","author":"Radford Alec","year":"2019","unstructured":"Alec Radford, Jeffrey Wu, Rewon Child, David Luan, Dario Amodei, Ilya Sutskever, 2019. Language models are unsupervised multitask learners. OpenAI blog 1, 8 (2019), 9."},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/1891903.1891912"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2010.5649430"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1177\/0278364919897133"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/DICTA.2015.7371296"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/ROMAN.2016.7745243"},{"key":"e_1_3_2_1_64_1","volume-title":"Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Thibaut Lavril, Gautier Izacard, Xavier Martinet, Marie-Anne Lachaux, Timoth\u00e9e Lacroix, Baptiste Rozi\u00e8re, Naman Goyal, Eric Hambro, Faisal Azhar, 2023. Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)."},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1162\/coli.2006.32.2.195"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.3115\/1706269.1706283"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.5555\/1708322.1708334"},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.1109\/FG.2017.145"},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300511"},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46475-6_5"}],"event":{"name":"ICMI '23: INTERNATIONAL CONFERENCE ON MULTIMODAL INTERACTION","location":"Paris France","acronym":"ICMI '23","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["International Cconference on Multimodal Interaction"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3610661.3616548","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3610661.3616548","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T19:31:22Z","timestamp":1755891082000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3610661.3616548"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,9]]},"references-count":71,"alternative-id":["10.1145\/3610661.3616548","10.1145\/3610661"],"URL":"https:\/\/doi.org\/10.1145\/3610661.3616548","relation":{},"subject":[],"published":{"date-parts":[[2023,10,9]]},"assertion":[{"value":"2023-10-09","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}