{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,8]],"date-time":"2026-03-08T02:35:50Z","timestamp":1772937350290,"version":"3.50.1"},"publisher-location":"Cham","reference-count":70,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031768262","type":"print"},{"value":"9783031768279","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-3-031-76827-9_1","type":"book-chapter","created":{"date-parts":[[2024,12,30]],"date-time":"2024-12-30T20:13:54Z","timestamp":1735589634000},"page":"3-22","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Multimodal Referring Expression Generation for\u00a0Human-Computer Interaction"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9253-416X","authenticated-orcid":false,"given":"Nada","family":"Alalyani","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7878-7227","authenticated-orcid":false,"given":"Nikhil","family":"Krishnaswamy","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,12,31]]},"reference":[{"key":"1_CR1","doi-asserted-by":"crossref","unstructured":"Alalyani, N., Krishnaswamy, N.: A methodology for evaluating multimodal referring expression generation for embodied virtual agents. In: Companion Publication of the 25th International Conference on Multimodal Interaction, pp. 164\u2013173 (2023)","DOI":"10.1145\/3610661.3616548"},{"key":"1_CR2","unstructured":"Alayrac, J.B., et al.: Flamingo: a visual language model for few-shot learning. In: Advances in Neural Information Processing Systems, vol. 35, pp. 23716\u201323736 (2022)"},{"key":"1_CR3","unstructured":"Alayrac, J.B., et al.: Self-supervised multimodal versatile networks. In: Advances in Neural Information Processing Systems, vol. 33, pp. 25\u201337 (2020)"},{"key":"1_CR4","unstructured":"Banerjee, S., Lavie, A.: Meteor: an automatic metric for MT evaluation with improved correlation with human judgments. In: Proceedings of the ACL Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation And\/Or Summarization, pp. 65\u201372 (2005)"},{"key":"1_CR5","doi-asserted-by":"crossref","unstructured":"Belz, A., Gatt, A.: Intrinsic vs. extrinsic evaluation measures for referring expression generation. In: Proceedings of ACL-08: HLT, Short Papers, pp. 197\u2013200 (2008)","DOI":"10.3115\/1557690.1557746"},{"key":"1_CR6","doi-asserted-by":"crossref","unstructured":"Bender, E.M., Koller, A.: Climbing towards NLU: on meaning, form, and understanding in the age of data. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, pp. 5185\u20135198 (2020)","DOI":"10.18653\/v1\/2020.acl-main.463"},{"key":"1_CR7","unstructured":"Brown, T., et al.: Language models are few-shot learners. In: Advances in Neural Information Processing Systems, vol. 33, pp. 1877\u20131901 (2020)"},{"key":"1_CR8","doi-asserted-by":"crossref","unstructured":"Chen, Y., et al.: Yourefit: embodied reference understanding with language and gesture. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1385\u20131395 (2021)","DOI":"10.1109\/ICCV48922.2021.00142"},{"key":"1_CR9","doi-asserted-by":"crossref","unstructured":"Chen, Z., Wang, P., Ma, L., Wong, K.Y.K., Wu, Q.: Cops-ref: a new dataset and task on compositional referring expression comprehension. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10086\u201310095 (2020)","DOI":"10.1109\/CVPR42600.2020.01010"},{"key":"1_CR10","doi-asserted-by":"crossref","unstructured":"De\u00a0Vries, H., Strub, F., Chandar, S., Pietquin, O., Larochelle, H., Courville, A.: Guesswhat?! Visual object discovery through multi-modal dialogue. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5503\u20135512 (2017)","DOI":"10.1109\/CVPR.2017.475"},{"issue":"3","key":"1_CR11","doi-asserted-by":"publisher","first-page":"297","DOI":"10.2307\/1932409","volume":"26","author":"LR Dice","year":"1945","unstructured":"Dice, L.R.: Measures of the amount of ecologic association between species. Ecology 26(3), 297\u2013302 (1945)","journal-title":"Ecology"},{"key":"1_CR12","doi-asserted-by":"crossref","unstructured":"Do\u011fan, F.I., Kalkan, S., Leite, I.: Learning to generate unambiguous spatial referring expressions for real-world environments. In: 2019 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 4992\u20134999. IEEE (2019)","DOI":"10.1109\/IROS40897.2019.8968510"},{"key":"1_CR13","doi-asserted-by":"crossref","unstructured":"Fang, R., Doering, M., Chai, J.Y.: Embodied collaborative referring expression generation in situated human-robot interaction. In: Proceedings of the Tenth Annual ACM\/IEEE International Conference on Human-Robot Interaction, pp. 271\u2013278 (2015)","DOI":"10.1145\/2696454.2696467"},{"key":"1_CR14","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"828","DOI":"10.1007\/978-3-540-73281-5_91","volume-title":"Universal Access in Human-Computer Interaction. Ambient Interaction","author":"ME Foster","year":"2007","unstructured":"Foster, M.E.: Enhancing human-computer interaction with embodied conversational agents. In: Stephanidis, C. (ed.) UAHCI 2007, Part II. LNCS, vol. 4555, pp. 828\u2013837. Springer, Heidelberg (2007). https:\/\/doi.org\/10.1007\/978-3-540-73281-5_91"},{"key":"1_CR15","doi-asserted-by":"crossref","unstructured":"Gatt, A., Belz, A., Kow, E.: The tuna-reg challenge 2009: overview and evaluation results. Assoc. Comput. Linguist. (2009)","DOI":"10.3115\/1610195.1610224"},{"issue":"4","key":"1_CR16","doi-asserted-by":"publisher","first-page":"423","DOI":"10.1007\/s10849-007-9047-0","volume":"16","author":"A Gatt","year":"2007","unstructured":"Gatt, A., Van Deemter, K.: Lexical choice and conceptual perspective in the generation of plural referring expressions. J. Logic Lang. Inform. 16(4), 423\u2013443 (2007)","journal-title":"J. Logic Lang. Inform."},{"issue":"11","key":"1_CR17","doi-asserted-by":"publisher","first-page":"419","DOI":"10.1016\/S1364-6613(99)01397-2","volume":"3","author":"S Goldin-Meadow","year":"1999","unstructured":"Goldin-Meadow, S.: The role of gesture in communication and thinking. Trends Cogn. Sci. 3(11), 419\u2013429 (1999)","journal-title":"Trends Cogn. Sci."},{"key":"1_CR18","doi-asserted-by":"publisher","first-page":"429","DOI":"10.1613\/jair.1327","volume":"21","author":"P Gorniak","year":"2004","unstructured":"Gorniak, P., Roy, D.: Grounded semantic composition for visual scenes. J. Artif. Intell. Res. 21, 429\u2013470 (2004)","journal-title":"J. Artif. Intell. Res."},{"key":"1_CR19","doi-asserted-by":"crossref","unstructured":"Han, L., Zheng, T., Xu, L., Fang, L.: OccuSeg: occupancy-aware 3D instance segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2940\u20132949 (2020)","DOI":"10.1109\/CVPR42600.2020.00301"},{"key":"1_CR20","unstructured":"Hu, E.J., et al.: LoRa: low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685 (2021)"},{"key":"1_CR21","unstructured":"Islam, M.M., Mirzaiee, R., Gladstone, A., Green, H., Iqbal, T.: Caesar: An embodied simulator for generating multimodal referring expression datasets. In: Advances in Neural Information Processing Systems, vol. 35, pp. 21001\u201321015 (2022)"},{"key":"1_CR22","doi-asserted-by":"publisher","first-page":"205","DOI":"10.1146\/annurev-control-070122-102501","volume":"6","author":"A Kalinowska","year":"2023","unstructured":"Kalinowska, A., Pilarski, P.M., Murphey, T.D.: Embodied communication: how robots and people communicate through physical interaction. Annu. Rev. Control Robot. Auton. Syst. 6, 205\u2013232 (2023)","journal-title":"Annu. Rev. Control Robot. Auton. Syst."},{"key":"1_CR23","unstructured":"Krahmer, E., van\u00a0der Sluis, I.: A new model for generating multimodal referring expressions. In: Proceedings of the ENLG, vol.\u00a03, pp. 47\u201354 (2003)"},{"key":"1_CR24","unstructured":"Kranstedt, A., Kopp, S., Wachsmuth, I.: MurML: a multimodal utterance representation markup language for conversational agents. In: AAMAS\u201902 Workshop Embodied Conversational Agents-Let\u2019s Specify and Evaluate Them! (2002)"},{"key":"1_CR25","unstructured":"Krishnaswamy, N., Alalyani, N.: Embodied multimodal agents to bridge the understanding gap. In: Proceedings of the First Workshop on Bridging Human\u2013Computer Interaction and Natural Language Processing, pp. 41\u201346 (2021)"},{"key":"1_CR26","unstructured":"Krishnaswamy, N., Alalyani, N.: Embodied multimodal agents to bridge the understanding gap. In: Proceedings of the First Workshop on Bridging Human\u2013Computer Interaction and Natural Language Processing, pp. 41\u201346. Association for Computational Linguistics, Online (2021)"},{"key":"1_CR27","doi-asserted-by":"crossref","unstructured":"Krishnaswamy, N., et al.: Diana\u2019s world: a situated multimodal interactive agent. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a034, pp. 13618\u201313619 (2020)","DOI":"10.1609\/aaai.v34i09.7096"},{"key":"1_CR28","unstructured":"Krishnaswamy, N., et\u00a0al.: Communicating and acting: understanding gesture in simulation semantics. In: Proceedings of the 12th International Conference on Computational Semantics (IWCS)-Short papers (2017)"},{"key":"1_CR29","unstructured":"Krishnaswamy, N., Pickard, W., Cates, B., Blanchard, N., Pustejovsky, J.: The voxworld platform for multimodal embodied agents. In: Proceedings of the Thirteenth Language Resources and Evaluation Conference, pp. 1529\u20131541 (2022)"},{"key":"1_CR30","unstructured":"Krishnaswamy, N., Pustejovsky, J.: Voxsim: a visual platform for modeling motion language. In: Proceedings of COLING 2016, the 26th International Conference on Computational Linguistics: System Demonstrations, pp. 54\u201358 (2016)"},{"key":"1_CR31","unstructured":"Krishnaswamy, N., Pustejovsky, J.: An evaluation framework for multimodal interaction. In: Proceedings of the Eleventh International Conference on Language Resources and Evaluation (LREC 2018) (2018)"},{"key":"1_CR32","doi-asserted-by":"crossref","unstructured":"Krishnaswamy, N., Pustejovsky, J.: Generating a novel dataset of multimodal referring expressions. In: Proceedings of the 13th International Conference on Computational Semantics-Short Papers, pp. 44\u201351 (2019)","DOI":"10.18653\/v1\/W19-0507"},{"key":"1_CR33","doi-asserted-by":"crossref","unstructured":"Krishnaswamy, N., Pustejovsky, J.: The role of embodiment and simulation in evaluating hci: Experiments and evaluation. In: International Conference on Human-Computer Interaction, pp. 220\u2013232 (2021)","DOI":"10.1007\/978-3-030-77817-0_17"},{"key":"1_CR34","doi-asserted-by":"publisher","DOI":"10.3389\/frai.2022.774752","volume":"5","author":"N Krishnaswamy","year":"2022","unstructured":"Krishnaswamy, N., Pustejovsky, J.: Affordance embeddings for situated language understanding. Front. Artif. Intell. 5, 774752 (2022)","journal-title":"Front. Artif. Intell."},{"key":"1_CR35","doi-asserted-by":"crossref","unstructured":"Kunze, L., Williams, T., Hawes, N., Scheutz, M.: Spatial referring expression generation for HRI: algorithms and evaluation framework. In: 2017 AAAI Fall Symposium Series (2017)","DOI":"10.18653\/v1\/W17-3511"},{"key":"1_CR36","unstructured":"Levenshtein, V.I., et\u00a0al.: Binary codes capable of correcting deletions, insertions, and reversals. In: Soviet Phys. Doklady, vol.\u00a010, pp. 707\u2013710. Soviet Union (1966)"},{"key":"1_CR37","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: Blip-2: bootstrapping language-image pre-training with frozen image encoders and large language models. arXiv preprint arXiv:2301.12597 (2023)"},{"key":"1_CR38","unstructured":"Li, L.H., Yatskar, M., Yin, D., Hsieh, C.J., Chang, K.W.: VisualBERT: a simple and performant baseline for vision and language. arXiv preprint arXiv:1908.03557 (2019)"},{"issue":"2","key":"1_CR39","doi-asserted-by":"publisher","first-page":"1494","DOI":"10.1109\/LRA.2022.3141150","volume":"7","author":"X Li","year":"2022","unstructured":"Li, X., Guo, D., Liu, H., Sun, F.: Reve-CE: remote embodied visual referring expression in continuous environment. IEEE Robot. Autom. Lett. 7(2), 1494\u20131501 (2022)","journal-title":"IEEE Robot. Autom. Lett."},{"key":"1_CR40","doi-asserted-by":"crossref","unstructured":"Lin, C.Y., Hovy, E.: Automatic evaluation of summaries using n-gram co-occurrence statistics. In: Proceedings of the 2003 Human Language Technology Conference of the North American Chapter of the Association for Computational Linguistics, pp. 150\u2013157 (2003)","DOI":"10.3115\/1073445.1073465"},{"key":"1_CR41","unstructured":"Lu, J., Batra, D., Parikh, D., Lee, S.: VilBERT: pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"key":"1_CR42","unstructured":"Ma, E.: NLP augmentation (2019). https:\/\/github.com\/makcedward\/nlpaug"},{"key":"1_CR43","unstructured":"Magassouba, A., Sugiura, K., Kawai, H.: Multimodal attention branch network for perspective-free sentence generation. In: Conference on Robot Learning, pp. 76\u201385. PMLR (2020)"},{"key":"1_CR44","doi-asserted-by":"crossref","unstructured":"Manning, C., Surdeanu, M., Bauer, J., Finkel, J., Bethard, S., McClosky, D.: The Stanford CoreNLP natural language processing toolkit. In: Proceedings of 52nd Annual Meeting of the Association for Computational Linguistics: System Demonstrations, pp. 55\u201360. Association for Computational Linguistics, Baltimore, Maryland (2014)","DOI":"10.3115\/v1\/P14-5010"},{"key":"1_CR45","doi-asserted-by":"crossref","unstructured":"Mao, J., Huang, J., Toshev, A., Camburu, O., Yuille, A.L., Murphy, K.: Generation and comprehension of unambiguous object descriptions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 11\u201320 (2016)","DOI":"10.1109\/CVPR.2016.9"},{"issue":"3","key":"1_CR46","doi-asserted-by":"publisher","first-page":"350","DOI":"10.1037\/0033-295X.92.3.350","volume":"92","author":"D McNeill","year":"1985","unstructured":"McNeill, D.: So you think gestures are nonverbal? Psychol. Rev. 92(3), 350 (1985)","journal-title":"Psychol. Rev."},{"key":"1_CR47","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.J.: Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, pp. 311\u2013318 (2002)","DOI":"10.3115\/1073083.1073135"},{"key":"1_CR48","unstructured":"Passonneau, R.: Measuring agreement on set-valued items (MASI) for semantic and pragmatic annotation (2006)"},{"key":"1_CR49","doi-asserted-by":"crossref","unstructured":"Pearson, K.: X. on the criterion that a given system of deviations from the probable in the case of a correlated system of variables is such that it can be reasonably supposed to have arisen from random sampling. The London, Edinburgh, and Dublin Philos. Mag. J. Sci. 50(302), 157\u2013175 (1900)","DOI":"10.1080\/14786440009463897"},{"key":"1_CR50","doi-asserted-by":"crossref","unstructured":"Pustejovsky, J., Krishnaswamy, N.: Embodied human-computer interactions through situated grounding. In: Proceedings of the 20th ACM International Conference on Intelligent Virtual Agents, pp.\u00a01\u20133 (2020)","DOI":"10.1145\/3383652.3423910"},{"issue":"3","key":"1_CR51","first-page":"17","volume":"61","author":"J Pustejovsky","year":"2020","unstructured":"Pustejovsky, J., Krishnaswamy, N.: Situated meaning in multimodal dialogue: human-robot and human-computer interactions. Traitement Automatique des Langues 61(3), 17\u201341 (2020)","journal-title":"Traitement Automatique des Langues"},{"issue":"3","key":"1_CR52","doi-asserted-by":"publisher","first-page":"307","DOI":"10.1007\/s13218-021-00727-5","volume":"35","author":"J Pustejovsky","year":"2021","unstructured":"Pustejovsky, J., Krishnaswamy, N.: Embodied human computer interaction. KI-K\u00fcnstliche Intelligenz 35(3), 307\u2013327 (2021)","journal-title":"KI-K\u00fcnstliche Intelligenz"},{"key":"1_CR53","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"137","DOI":"10.1007\/978-3-031-05311-5_9","volume-title":"Human-Computer Interaction. Theoretical Approaches and Design Methods - HCII 2022","author":"J Pustejovsky","year":"2022","unstructured":"Pustejovsky, J., Krishnaswamy, N.: Multimodal semantics for affordances and actions. In: Kurosu, M. (ed.) HCII 2022. LNCS, vol. 13302, pp. 137\u2013160. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-05311-5_9"},{"key":"1_CR54","doi-asserted-by":"crossref","unstructured":"Qi, Y., et al.: Reverie: remote embodied visual referring expression in real indoor environments. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9982\u20139991 (2020)","DOI":"10.1109\/CVPR42600.2020.01000"},{"key":"1_CR55","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"1_CR56","doi-asserted-by":"crossref","unstructured":"Schauerte, B., Fink, G.A.: Focusing computational visual attention in multi-modal human-robot interaction. In: International Conference on Multimodal Interfaces and the Workshop on Machine Learning for Multimodal Interaction, pp.\u00a01\u20138 (2010)","DOI":"10.1145\/1891903.1891912"},{"key":"1_CR57","doi-asserted-by":"crossref","unstructured":"Schauerte, B., Richarz, J., Fink, G.A.: Saliency-based identification and recognition of pointed-at objects. In: 2010 IEEE\/RSJ International Conference on Intelligent Robots and Systems, pp. 4638\u20134643. IEEE (2010)","DOI":"10.1109\/IROS.2010.5649430"},{"issue":"2\u20133","key":"1_CR58","doi-asserted-by":"publisher","first-page":"217","DOI":"10.1177\/0278364919897133","volume":"39","author":"M Shridhar","year":"2020","unstructured":"Shridhar, M., Mittal, D., Hsu, D.: Ingress: interactive visual grounding of referring expressions. Int. J. Robot. Res. 39(2\u20133), 217\u2013232 (2020)","journal-title":"Int. J. Robot. Res."},{"key":"1_CR59","doi-asserted-by":"crossref","unstructured":"Shukla, D., Erkent, O., Piater, J.: Probabilistic detection of pointing directions for human-robot interaction. In: 2015 International Conference on Digital Image Computing: Techniques and Applications (DICTA), pp.\u00a01\u20138. IEEE (2015)","DOI":"10.1109\/DICTA.2015.7371296"},{"key":"1_CR60","doi-asserted-by":"crossref","unstructured":"Shukla, D., Erkent, \u00d6., Piater, J.: A multi-view hand gesture rgb-d dataset for human-robot interaction scenarios. In: 2016 25th IEEE International Symposium on Robot and Human Interactive Communication (RO-MAN), pp. 1084\u20131091. IEEE (2016)","DOI":"10.1109\/ROMAN.2016.7745243"},{"key":"1_CR61","unstructured":"Taori, R., et al.: Stanford alpaca: an instruction-following llama model (2023). https:\/\/github.com\/tatsu-lab\/stanford_alpaca"},{"key":"1_CR62","unstructured":"Touvron, H., et\u00a0al.: Llama: open and efficient foundation language models (2023). arXiv preprint arXiv:2302.13971 (2023)"},{"issue":"2","key":"1_CR63","doi-asserted-by":"publisher","first-page":"195","DOI":"10.1162\/coli.2006.32.2.195","volume":"32","author":"K Van Deemter","year":"2006","unstructured":"Van Deemter, K.: Generating referring expressions that involve gradable properties. Comput. Linguist. 32(2), 195\u2013222 (2006)","journal-title":"Comput. Linguist."},{"key":"1_CR64","doi-asserted-by":"crossref","unstructured":"Viethen, J., Dale, R.: Algorithms for generating referring expressions: do they do what people do? In: Proceedings of the Fourth International Natural Language Generation Conference, pp. 63\u201370 (2006)","DOI":"10.3115\/1706269.1706283"},{"key":"1_CR65","doi-asserted-by":"crossref","unstructured":"Viethen, J., Dale, R.: The use of spatial relations in referring expression generation. In: Proceedings of the Fifth International Natural Language Generation Conference, pp. 59\u201367 (2008)","DOI":"10.3115\/1708322.1708334"},{"key":"1_CR66","doi-asserted-by":"crossref","unstructured":"Wang, I., Smith, J., Ruiz, J.: Exploring virtual agents for augmented reality. In: Proceedings of the 2019 CHI Conference on Human Factors in Computing Systems, pp. 1\u201312 (2019)","DOI":"10.1145\/3290605.3300511"},{"key":"1_CR67","unstructured":"Wang, Z., Yu, J., Yu, A.W., Dai, Z., Tsvetkov, Y., Cao, Y.: SimVLM: simple visual language model pretraining with weak supervision. arXiv preprint arXiv:2108.10904 (2021)"},{"key":"1_CR68","unstructured":"Xu, M., et\u00a0al.: A survey of resource-efficient LLM and multimodal foundation models. arXiv preprint arXiv:2401.08092 (2024)"},{"key":"1_CR69","unstructured":"Zhang, T., Kishore, V., Wu, F., Weinberger, K.Q., Artzi, Y.: Bertscore: evaluating text generation with BERT. In: International Conference on Learning Representations (2019)"},{"key":"1_CR70","doi-asserted-by":"crossref","unstructured":"Zhu, L., Yang, Y.: ActBERT: learning global-local video-text representations. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8746\u20138755 (2020)","DOI":"10.1109\/CVPR42600.2020.00877"}],"container-title":["Lecture Notes in Computer Science","HCI International 2024 \u2013 Late Breaking Papers"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-76827-9_1","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,30]],"date-time":"2024-12-30T21:06:59Z","timestamp":1735592819000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-76827-9_1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9783031768262","9783031768279"],"references-count":70,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-76827-9_1","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"31 December 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"HCII","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Human-Computer Interaction","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Washington DC","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"USA","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 June 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 July 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"hcii2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/2024.hci.international\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}