{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,27]],"date-time":"2026-03-27T01:12:08Z","timestamp":1774573928703,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":18,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,3,11]],"date-time":"2024-03-11T00:00:00Z","timestamp":1710115200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100006374","name":"H2020 Marie Sk\u0142odowska-Curie Actions","doi-asserted-by":"publisher","award":["801342"],"award-info":[{"award-number":["801342"]}],"id":[{"id":"10.13039\/501100006374","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100006374","name":"HORIZON EUROPE Framework Programme","doi-asserted-by":"publisher","award":["871245"],"award-info":[{"award-number":["871245"]}],"id":[{"id":"10.13039\/501100006374","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,3,11]]},"DOI":"10.1145\/3610977.3637473","type":"proceedings-article","created":{"date-parts":[[2024,3,10]],"date-time":"2024-03-10T00:19:00Z","timestamp":1710029940000},"page":"865-869","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["Dataset and Evaluation of Automatic Speech Recognition for Multi-lingual Intent Recognition on Social Robots"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6641-6450","authenticated-orcid":false,"given":"Antonio","family":"Andriella","sequence":"first","affiliation":[{"name":"Pal Robotics, Barcelona, Spain"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8295-6932","authenticated-orcid":false,"given":"Raquel","family":"Ros","sequence":"additional","affiliation":[{"name":"Pal Robotics, Barcelona, Spain"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-6116-2869","authenticated-orcid":false,"given":"Yoav","family":"Ellinson","sequence":"additional","affiliation":[{"name":"Bar-Ilan University, Tel-Aviv, Israel"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2885-170X","authenticated-orcid":false,"given":"Sharon","family":"Gannot","sequence":"additional","affiliation":[{"name":"Bar-Ilan University, Tel-Aviv, Israel"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3391-8876","authenticated-orcid":false,"given":"S\u00e9verin","family":"Lemaignan","sequence":"additional","affiliation":[{"name":"PAL Robotics, Barcelona, Spain"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,3,11]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Online. https:\/\/alphacephei.com\/vosk\/."},{"key":"e_1_3_2_1_2_1","unstructured":"Online. https:\/\/cloud.google.com\/speech-to-text\/?hl=en."},{"key":"e_1_3_2_1_3_1","unstructured":"Online. https:\/\/github.com\/openai\/whisper."},{"key":"e_1_3_2_1_4_1","unstructured":"Online. https:\/\/www.nvidia.com\/en-us\/ai-data-science\/products\/riva\/."},{"key":"e_1_3_2_1_5_1","unstructured":"Online. https:\/\/rasa.com\/."},{"key":"e_1_3_2_1_6_1","unstructured":"Online. https:\/\/rasa.com\/docs\/rasa\/glossary\/#intent."},{"key":"e_1_3_2_1_7_1","unstructured":"Online. https:\/\/rasa.com\/docs\/rasa\/glossary\/#entity."},{"key":"e_1_3_2_1_8_1","unstructured":"Online. https:\/\/wiki.seeedstudio.com\/ReSpeaker_Mic_Array_v2.0\/."},{"key":"e_1_3_2_1_9_1","unstructured":"Online. https:\/\/pypi.org\/project\/jiwer."},{"key":"e_1_3_2_1_10_1","volume-title":"Interpreting and Explaining Deep Neural Networks for Classification of Audio Signals. CoRR abs\/1807.03418","author":"Becker S\u00f6ren","year":"2018","unstructured":"S\u00f6ren Becker, Marcel Ackermann, Sebastian Lapuschkin, Klaus-Robert M\u00fcller, and Wojciech Samek. 2018. Interpreting and Explaining Deep Neural Networks for Classification of Audio Signals. CoRR abs\/1807.03418 (2018). arXiv:1807.03418"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/RO-MAN47096.2020.9223470"},{"key":"e_1_3_2_1_12_1","unstructured":"Lingyun Feng Jianwei Yu Deng Cai Songxiang Liu Haitao Zheng and Yan Wang. 2022. ASR-GLUE: A New Multi-task Benchmark for ASR-Robust Natural Language Understanding. arXiv:2108.13048 [cs.CL]"},{"key":"e_1_3_2_1_13_1","volume-title":"Conformer: Convolution-augmented Transformer for Speech Recognition. arXiv:2005.08100 [eess.AS]","author":"Gulati Anmol","year":"2020","unstructured":"Anmol Gulati, James Qin, Chung-Cheng Chiu, Niki Parmar, Yu Zhang, Jiahui Yu, Wei Han, Shibo Wang, Zhengdong Zhang, Yonghui Wu, and Ruoming Pang. 2020. Conformer: Convolution-augmented Transformer for Speech Recognition. arXiv:2005.08100 [eess.AS]"},{"key":"e_1_3_2_1_14_1","unstructured":"Zohar Jackson C\u00e9sar Souza Jason Flaks Yuxin Pan Hereman Nicolas and Adhish Thite. 2018. Jakobovski\/free-spoken-digit-dataset: v1.0.8. https:\/\/doi.org\/10.5281\/ zenodo.1342401"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3568294.3580041"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-020--10073--7"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.5220\/0007690902550265"},{"key":"e_1_3_2_1_18_1","volume-title":"Automatic Speech Recognition for Indoor HRI Scenarios. ACM Transactions on Human-Robot Interaction 10 (03","author":"Novoa-Ilic Jos\u00e9","year":"2021","unstructured":"Jos\u00e9 Novoa-Ilic, Rodrigo Mahu, Jorge Wuth, Juan Escudero, Josu\u00e9 Fredes, and Nestor Yoma. 2021. Automatic Speech Recognition for Indoor HRI Scenarios. ACM Transactions on Human-Robot Interaction 10 (03 2021), 1--30. https:\/\/doi. org\/10.1145\/3442629"}],"event":{"name":"HRI '24: ACM\/IEEE International Conference on Human-Robot Interaction","location":"Boulder CO USA","acronym":"HRI '24","sponsor":["SIGAI ACM Special Interest Group on Artificial Intelligence","SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 2024 ACM\/IEEE International Conference on Human-Robot Interaction"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3610977.3637473","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3610977.3637473","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,28]],"date-time":"2025-08-28T16:34:27Z","timestamp":1756398867000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3610977.3637473"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,3,11]]},"references-count":18,"alternative-id":["10.1145\/3610977.3637473","10.1145\/3610977"],"URL":"https:\/\/doi.org\/10.1145\/3610977.3637473","relation":{},"subject":[],"published":{"date-parts":[[2024,3,11]]},"assertion":[{"value":"2024-03-11","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}