{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T00:00:16Z","timestamp":1782259216365,"version":"3.54.5"},"reference-count":38,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,8,28]],"date-time":"2023-08-28T00:00:00Z","timestamp":1693180800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,8,28]],"date-time":"2023-08-28T00:00:00Z","timestamp":1693180800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001321","name":"National Research Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001321","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,8,28]]},"DOI":"10.1109\/ro-man57019.2023.10309475","type":"proceedings-article","created":{"date-parts":[[2023,11,13]],"date-time":"2023-11-13T19:07:48Z","timestamp":1699902468000},"page":"1648-1654","source":"Crossref","is-referenced-by-count":2,"title":["SGGNet<sup>2<\/sup>: Speech-Scene Graph Grounding Network for Speech-guided Navigation"],"prefix":"10.1109","author":[{"given":"Dohyun","family":"Kim","sequence":"first","affiliation":[{"name":"Korea Advanced Institute of Science and Technology,Robust Intelligence and Robotics (RIRO) laboratory,Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yeseung","family":"Kim","sequence":"additional","affiliation":[{"name":"Korea Advanced Institute of Science and Technology,Robust Intelligence and Robotics (RIRO) laboratory,Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jaehwi","family":"Jang","sequence":"additional","affiliation":[{"name":"Korea Advanced Institute of Science and Technology,Robust Intelligence and Robotics (RIRO) laboratory,Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Minjae","family":"Song","sequence":"additional","affiliation":[{"name":"Korea Advanced Institute of Science and Technology,Robust Intelligence and Robotics (RIRO) laboratory,Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Woojin","family":"Choi","sequence":"additional","affiliation":[{"name":"Korea Advanced Institute of Science and Technology,Robust Intelligence and Robotics (RIRO) laboratory,Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Daehyung","family":"Park","sequence":"additional","affiliation":[{"name":"Korea Advanced Institute of Science and Technology,Robust Intelligence and Robotics (RIRO) laboratory,Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2018.2801475"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2015.7353563"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0221854"},{"key":"ref4","article-title":"Grounding robot plans from natural language instructions with incomplete world knowledge","volume-title":"Proc. Conf. on Robot Learning","author":"Nyga"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.55417\/fr.2022017"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-26889-2_14"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/k19-1040"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-33950-0_39"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v25i1.7979"},{"key":"ref10","article-title":"Do as i can, not as i say: Grounding language in robotic affordances","volume-title":"Proc. Conf. on Robot Learning","author":"Ichter"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472621"},{"key":"ref12","article-title":"Automatic Speech Recognition (ASR) \u2014 NVIDIA NeMo","year":"2023"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1669"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_25"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58565-5_13"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA40945.2020.9197315"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2014.6907841"},{"key":"ref18","article-title":"Improving language understanding by generative pre-training","author":"Radford","year":"2018","journal-title":"OpenAI"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.18653\/vl\/N19-142"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/IROS47612.2022.9981810"},{"key":"ref21","article-title":"Learning multi-modal grounded linguistic semantics by playing \u201ci spy","volume-title":"Proc. Int\u2019l Joint Conf. on Artificial Intelligence","author":"Thomason"},{"key":"ref22","article-title":"Speech to text \u2014 Microsoft Azure","year":"2023"},{"key":"ref23","article-title":"Speech-to-Text: Automatic speech recognition-Google Cloud","year":"2023"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2020-3015"},{"key":"ref25","article-title":"Distilkobert: Distillation of kobert","year":"2023"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.3390\/app10196936"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1016\/j.simpa.2021.100054"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.91"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1007\/s10514-019-09883-y"},{"key":"ref32","article-title":"Graph attention networks","volume-title":"Proc. Int\u2019l Conf. on Learning Representation","author":"Veli\u010dkovi\u0107"},{"key":"ref33","article-title":"Graphvqa: Language-guided graph neural networks for scene graph question answering","volume-title":"Proc. Conf. of the North American Chapter of the Assoc. for Comp. Linguistics: Human Language Technologies","author":"Liang"},{"key":"ref34","article-title":"Decoupled weight decay regularization","volume-title":"Proc. Int\u2019l Conf. on Learning Representation","author":"Loshchilov"},{"key":"ref35","article-title":"Sgdr: Stochastic gradient descent with warm restarts","volume-title":"Proc. Int\u2019l Conf. on Learning Representation","author":"Loshchilov"},{"key":"ref36","article-title":"Gpt-4 technical report","year":"2023"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1145\/3209978.3210208"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/TRO.2022.3141876"}],"event":{"name":"2023 32nd IEEE International Conference on Robot and Human Interactive Communication (RO-MAN)","location":"Busan, Korea, Republic of","start":{"date-parts":[[2023,8,28]]},"end":{"date-parts":[[2023,8,31]]}},"container-title":["2023 32nd IEEE International Conference on Robot and Human Interactive Communication (RO-MAN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10309296\/10309265\/10309475.pdf?arnumber=10309475","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,2]],"date-time":"2024-03-02T13:53:41Z","timestamp":1709387621000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10309475\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,8,28]]},"references-count":38,"URL":"https:\/\/doi.org\/10.1109\/ro-man57019.2023.10309475","relation":{},"subject":[],"published":{"date-parts":[[2023,8,28]]}}}