{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,29]],"date-time":"2026-07-29T20:29:46Z","timestamp":1785356986815,"version":"3.55.0"},"reference-count":50,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2025,1,17]],"date-time":"2025-01-17T00:00:00Z","timestamp":1737072000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,17]],"date-time":"2025-01-17T00:00:00Z","timestamp":1737072000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Intel Serv Robotics"],"published-print":{"date-parts":[[2025,3]]},"DOI":"10.1007\/s11370-024-00570-1","type":"journal-article","created":{"date-parts":[[2025,1,17]],"date-time":"2025-01-17T20:38:46Z","timestamp":1737146326000},"page":"261-277","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":22,"title":["Human\u2013robot interaction through joint robot planning with large language models"],"prefix":"10.1007","volume":"18","author":[{"given":"Kosi","family":"Asuzu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Harjinder","family":"Singh","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9995-3180","authenticated-orcid":false,"given":"Moad","family":"Idrissi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,1,17]]},"reference":[{"key":"570_CR1","unstructured":"Ahn M, Brohan A, Brown N et al (2022) Do as I can, not as I say: grounding language in robotic affordances. arXiv preprint arXiv:220401691"},{"key":"570_CR2","doi-asserted-by":"crossref","unstructured":"Besta M, Blach N, Kubicek A et al (2024) Graph of thoughts: solving elaborate problems with large language models. In: Proceedings of the AAAI conference on artificial intelligence, pp 17682\u201317690","DOI":"10.1609\/aaai.v38i16.29720"},{"key":"570_CR3","doi-asserted-by":"crossref","unstructured":"Bragan\u00b8ca S, Costa E, Castellucci I et al (2019) A brief overview of the use of collaborative robots in industry 4.0: human role and safety. Occupational and environmental safety and health pp 641\u2013650","DOI":"10.1007\/978-3-030-14730-3_68"},{"key":"570_CR4","unstructured":"Brohan A, Brown N, Carbajal J et al (2023) Rt-2: Vision-language-action models transfer web knowledge to robotic control. arXiv preprint arXiv:230715818"},{"key":"570_CR5","first-page":"1877","volume":"33","author":"T Brown","year":"2020","unstructured":"Brown T, Mann B, Ryder N et al (2020) Language models are few-shot learners. Adv Neural Inf Process Syst 33:1877\u20131901","journal-title":"Adv Neural Inf Process Syst"},{"key":"570_CR6","unstructured":"Chen L, Zhang Y, Ren S, et al (2023) Towards end-to-end embodied decision making via multi-modal large language model: Explorations with gpt4-vision and beyond. arXiv preprint arXiv:231002071"},{"key":"570_CR7","unstructured":"Clabaugh C, Tsiakas K, Mataric M (2017) Predicting preschool mathematics performance of children with a socially assistive robot tutor. In: Proceedings of the synergies between learning and interaction workshop@ IROS, Vancouver, BC, Canada, pp 24\u201328"},{"key":"570_CR8","doi-asserted-by":"publisher","first-page":"104335","DOI":"10.1016\/j.robot.2022.104335","volume":"161","author":"A Dahiya","year":"2023","unstructured":"Dahiya A, Aroyo AM, Dautenhahn K et al (2023) A survey of multi-agent human\u2013robot interaction systems. Robot Auton Syst 161:104335","journal-title":"Robot Auton Syst"},{"key":"570_CR9","unstructured":"Ding Y, Zhang X, Amiri S et al (2022) Robot task planning and situation handling in open worlds. arXiv preprint arXiv:221001287"},{"key":"570_CR10","doi-asserted-by":"crossref","unstructured":"Ding Y, Zhang X, Paxton C et al (2023) Task and motion planning with large language models for object rearrangement. In: 2023 IEEE\/RSJ international conference on intelligent robots and systems (IROS), IEEE pp 2086\u20132092","DOI":"10.1109\/IROS55552.2023.10342169"},{"key":"570_CR11","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A et al (2020) An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:201011929"},{"key":"570_CR12","unstructured":"Driess D, Xia F, Sajjadi MS et al (2023) Palm-e: an embodied multimodal language model. arXiv preprint arXiv:230303378"},{"key":"570_CR13","unstructured":"Gu X, Lin TY, Kuo W et al (2021) Open-vocabulary object detection via vision and language knowledge distillation. arXiv preprint arXiv:210413921"},{"key":"570_CR14","doi-asserted-by":"crossref","unstructured":"Hadi MU, Qureshi R, Shah A, et al (2023) Large language models: a comprehensive survey of its applications, challenges, limitations, and future prospects. Authorea preprints","DOI":"10.36227\/techrxiv.23589741.v2"},{"key":"570_CR15","doi-asserted-by":"publisher","unstructured":"Huang S, Jiang Z, Dong H et al (2023) Instruct2act: mapping multi-modality instructions to robotic actions with large language model. arXiv preprint arXiv:230511176https:\/\/doi.org\/10.48550\/arxiv.2305.11176","DOI":"10.48550\/arxiv.2305.11176"},{"key":"570_CR16","unstructured":"Huang W, Abbeel P, Pathak D et al (2022) Language models as zero-shot planners: Extracting actionable knowledge for embodied agents. In: International conference on machine learning, PMLR, pp 9118\u20139147"},{"key":"570_CR17","unstructured":"Intel (n.d.) Intel\u00ae realsense\u2122 depth camera d435i. https:\/\/www.intelrealsense.com\/depth-camera-d435i\/. Accessed 10 July 2024"},{"key":"570_CR18","doi-asserted-by":"publisher","first-page":"222","DOI":"10.1016\/j.cogr.2022.10.001","volume":"2","author":"M Javaid","year":"2022","unstructured":"Javaid M, Haleem A, Singh RP et al (2022) Significant applications of cobots in the field of manufacturing. Cognit Robot 2:222\u2013233","journal-title":"Cognit Robot"},{"key":"570_CR19","unstructured":"Jiang Y, Gupta A, Zhang Z et al (2022) Vima: General robot manipulation with multimodal prompts, 2(3):6. arXiv preprint arXiv:221003094"},{"issue":"1","key":"570_CR20","doi-asserted-by":"publisher","first-page":"266","DOI":"10.1109\/TSMC.2020.3018325","volume":"51","author":"AI K\u00e1roly","year":"2020","unstructured":"K\u00e1roly AI, Galambos P, Kuti J et al (2020) Deep learning in robotics: Survey on model structures and training strategies. IEEE Trans Syst Man Cybern Syst 51(1):266\u2013279","journal-title":"IEEE Trans Syst Man Cybern Syst"},{"key":"570_CR21","doi-asserted-by":"crossref","unstructured":"Kirillov A, Mintun E, Ravi N et al (2023) Segment anything. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 4015\u20134026","DOI":"10.1109\/ICCV51070.2023.00371"},{"issue":"2","key":"570_CR22","first-page":"13","volume":"3","author":"M Knudsen","year":"2020","unstructured":"Knudsen M, Kaivo-Oja J (2020) Collaborative robots: frontiers of current literature. J Intell Syst: Theory Appl 3(2):13\u201320","journal-title":"J Intell Syst: Theory Appl"},{"key":"570_CR23","first-page":"22199","volume":"35","author":"T Kojima","year":"2022","unstructured":"Kojima T, Gu SS, Reid M et al (2022) Large language models are zero-shot reasoners. Adv Neural Inf Process Syst 35:22199\u201322213","journal-title":"Adv Neural Inf Process Syst"},{"issue":"Suppl 1","key":"570_CR24","doi-asserted-by":"publisher","first-page":"S85","DOI":"10.1134\/S1064562422060138","volume":"106","author":"AK Kovalev","year":"2022","unstructured":"Kovalev AK, Panov AI (2022) Application of pretrained large language models in embodied artificial intelligence. Dokl Math 106(Suppl 1):S85\u2013S90","journal-title":"Dokl Math"},{"key":"570_CR25","doi-asserted-by":"crossref","unstructured":"Liang J, Huang W, Xia F et al (2023) Code as policies: Language model programs for embodied control. In: 2023 IEEE international conference on robotics and automation (ICRA), IEEE, pp 9493\u20139500","DOI":"10.1109\/ICRA48891.2023.10160591"},{"key":"570_CR26","doi-asserted-by":"publisher","first-page":"111","DOI":"10.1016\/j.aiopen.2022.10.001","volume":"3","author":"T Lin","year":"2022","unstructured":"Lin T, Wang Y, Liu X et al (2022) A survey of transformers. AI open 3:111\u2013132","journal-title":"AI open"},{"key":"570_CR27","unstructured":"Liu B, Jiang Y, Zhang X et al (2023) LLM+ p: empowering large language models with optimal planning proficiency. arXiv preprint arXiv:230411477"},{"key":"570_CR28","doi-asserted-by":"crossref","unstructured":"Lykov A, Tsetserukou D (2023) LLM-brain: AI-driven fast generation of robot behaviour tree based on large language model. arXiv preprint arXiv:230519352","DOI":"10.1109\/FLLM63129.2024.10852491"},{"key":"570_CR29","unstructured":"Niryo (n.d.) Ned2. https:\/\/niryo.com\/products-cobots\/robot-ned-2\/. Accessed 10 July 2024"},{"key":"570_CR30","doi-asserted-by":"crossref","unstructured":"Park JS, O\u2019Brien J, Cai CJ et al (2023) Generative agents: Interactive simulacra of human behavior. In: Proceedings of the 36th annual ACM symposium on user interface software and technology, pp 1\u201322","DOI":"10.1145\/3586183.3606763"},{"key":"570_CR31","unstructured":"Radford A, Kim JW, Hallacy C et al (2021) Learning transferable visual models from natural language supervision. In: International conference on machine learning, PMLR, pp 8748\u20138763"},{"key":"570_CR32","unstructured":"Raman SS, Cohen V, Paulius D et al (2022) Cape: Corrective actions from precondition errors using large language models. arXiv preprint arXiv:221109935"},{"key":"570_CR33","unstructured":"Sharkawy AN (2021) Human\u2013robot interaction: applications. arXiv preprint arXiv:210200928"},{"key":"570_CR34","doi-asserted-by":"crossref","unstructured":"Singh I, Blukis V, Mousavian A et al (2023) Progprompt: generating situated robot task plans using large language models. In: 2023 IEEE international conference on robotics and automation (ICRA), IEEE, pp 11523\u201311530","DOI":"10.1109\/ICRA48891.2023.10161317"},{"key":"570_CR35","unstructured":"Sun H, Zhuang Y, Kong L et al (2024) Adaplanner: adaptive planning from feedback with language models. In: Advances in neural information processing systems 36"},{"key":"570_CR36","doi-asserted-by":"publisher","first-page":"159","DOI":"10.1016\/j.cogr.2021.08.001","volume":"1","author":"C Tan","year":"2021","unstructured":"Tan C, Xu X, Shen F (2021) A survey of zero shot detection: methods and applications. Cognit Robot 1:159\u2013167","journal-title":"Cognit Robot"},{"key":"570_CR37","unstructured":"Vaswani A, Shazeer N, Parmar N et al (2017) Attention is all you need. In: Advances in neural information processing systems 30"},{"issue":"4","key":"570_CR38","doi-asserted-by":"publisher","DOI":"10.1115\/1.4046238","volume":"143","author":"F Vicentini","year":"2021","unstructured":"Vicentini F (2021) Collaborative robotics: a survey. J Mech Des 143(4):040802","journal-title":"J Mech Des"},{"key":"570_CR39","first-page":"24824","volume":"35","author":"J Wei","year":"2022","unstructured":"Wei J, Wang X, Schuurmans D et al (2022) Chain-of-thought prompting elicits reasoning in large language models. Adv Neural Inf Process Syst 35:24824\u201324837","journal-title":"Adv Neural Inf Process Syst"},{"issue":"8","key":"570_CR40","doi-asserted-by":"publisher","first-page":"1087","DOI":"10.1007\/s10514-023-10139-z","volume":"47","author":"J Wu","year":"2023","unstructured":"Wu J, Antonova R, Kan A et al (2023) Tidybot: personalized robot assistance with large language models. Auton Robot 47(8):1087\u20131102","journal-title":"Auton Robot"},{"key":"570_CR41","unstructured":"Xie Y, Yu C, Zhu T et al (2023) Translating natural language to planning goals with large-language models. arXiv preprint arXiv:230205128"},{"key":"570_CR42","unstructured":"Yao S, Yu D, Zhao J et al (2024) Tree of thoughts: Deliberate problem solving with large language models. In: Advances in neural information processing systems 36"},{"key":"570_CR43","doi-asserted-by":"crossref","unstructured":"Ye F, Zhang S, Wang P et al (2021) A survey of deep reinforcement learning algorithms for motion planning and control of autonomous vehicles. In: 2021 IEEE intelligent vehicles symposium (IV), IEEE, pp 1073\u20131080","DOI":"10.1109\/IV48863.2021.9575880"},{"key":"570_CR44","unstructured":"Ye J, Chen X, Xu N et al (2023) A comprehensive capability analysis of gpt-3 and gpt-3.5 series models. arXiv preprint arXiv:230310420"},{"key":"570_CR45","unstructured":"Yu W, Gileadi N, Fu C et al (2023) Language to rewards for robotic skill synthesis. arXiv preprint arXiv:230608647"},{"key":"570_CR46","doi-asserted-by":"crossref","unstructured":"Zhang B, Soh H (2023) Large language models as zero-shot human models for human\u2013robot interaction. In: 2023 IEEE\/RSJ international conference on intelligent robots and systems (IROS), IEEE, pp 7961\u20137968","DOI":"10.1109\/IROS55552.2023.10341488"},{"key":"570_CR47","unstructured":"Zhang H, Du W, Shan J et al (2023) Building cooperative embodied agents modularly with large language models. arXiv preprint arXiv:230702485"},{"key":"570_CR48","doi-asserted-by":"crossref","unstructured":"Zhao X, Li M, Weber C et al (2023) Chat with the environment: interactive multimodal perception using large language models. In: 2023 IEEE\/RSJ international conference on intelligent robots and systems (IROS), IEEE, pp 3590\u20133596","DOI":"10.1109\/IROS55552.2023.10342363"},{"key":"570_CR49","unstructured":"Zhao Z, Lee WS, Hsu D (2024) Large language models as commonsense knowledge for large-scale task planning. In: Advances in neural information processing systems 36"},{"issue":"4","key":"570_CR50","doi-asserted-by":"publisher","first-page":"998","DOI":"10.1109\/TCSVT.2019.2899569","volume":"30","author":"P Zhu","year":"2019","unstructured":"Zhu P, Wang H, Saligrama V (2019) Zero shot detection. IEEE Trans Circuits Syst Video Technol 30(4):998\u20131010","journal-title":"IEEE Trans Circuits Syst Video Technol"}],"container-title":["Intelligent Service Robotics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11370-024-00570-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11370-024-00570-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11370-024-00570-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,5]],"date-time":"2025-04-05T09:28:53Z","timestamp":1743845333000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11370-024-00570-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,1,17]]},"references-count":50,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2025,3]]}},"alternative-id":["570"],"URL":"https:\/\/doi.org\/10.1007\/s11370-024-00570-1","relation":{},"ISSN":["1861-2776","1861-2784"],"issn-type":[{"value":"1861-2776","type":"print"},{"value":"1861-2784","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,1,17]]},"assertion":[{"value":"18 January 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 October 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 January 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"There are no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}},{"value":"Not applicable.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}}]}}