{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T14:46:57Z","timestamp":1782312417903,"version":"3.54.5"},"publisher-location":"Cham","reference-count":15,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032058911","type":"print"},{"value":"9783032058928","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T00:00:00Z","timestamp":1759276800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T00:00:00Z","timestamp":1759276800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-05892-8_1","type":"book-chapter","created":{"date-parts":[[2025,9,30]],"date-time":"2025-09-30T23:18:07Z","timestamp":1759274287000},"page":"1-12","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Development of a Multimodal Dialogue System Integrating Speech Recognition, Knowledge Reasoning, and TTS via Multimodal Large Language Models"],"prefix":"10.1007","author":[{"given":"Zhiyi","family":"Gao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Akio","family":"Doi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Toru","family":"Katoh","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Meguru","family":"Yamashita","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hiroki","family":"Takahashi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,10,1]]},"reference":[{"key":"1_CR1","unstructured":"Brown, T., et al.: Language models are few-shot learners. In: Advances in Neural Information Processing Systems, vol. 33, pp. 1877\u20131901 (2020)"},{"key":"1_CR2","unstructured":"Zellers, R., et al.: Defending against neural fake news. NeurIPS (2019)"},{"key":"1_CR3","unstructured":"LangChain. LangChain Documentation (2023). https:\/\/docs.langchain.com"},{"key":"1_CR4","unstructured":"Pavlov, M., et al.: Multimodal interfaces for human-AI collaboration. In: Proceedings of the ACM on Human-Computer Interaction (IUI) (2022)"},{"key":"1_CR5","unstructured":"DeepSeek. DeepSeek Models Technical Overview (2024). https:\/\/github.com\/deepseek-ai"},{"key":"1_CR6","unstructured":"Ollama. Run open LLMs locally (2023). https:\/\/ollama.com"},{"key":"1_CR7","unstructured":"Google Cloud. Speech-to-Text Documentation (2023). https:\/\/cloud.google.com\/speech-to-text\/docs"},{"key":"1_CR8","unstructured":"Google. gTTS: Python Text-to-Speech (2023). https:\/\/pypi.org\/project\/gTTS\/"},{"key":"1_CR9","unstructured":"Microsoft Azure. Speech service - Text to speech (2023). https:\/\/learn.microsoft.com\/en-us\/azure\/cognitive-services\/speech-service\/text-to-speech"},{"key":"1_CR10","unstructured":"Xu, H., et al.: A comparative study of open-source text-to-speech synthesis systems. Proc. Interspeech (2021)"},{"key":"1_CR11","unstructured":"Kim, J., et al.: Real-time multilingual voice interfaces using edge devices. IEEE Access (2023)"},{"key":"1_CR12","unstructured":"Shneiderman, B.: Human-centered AI. ACM Interact. (2020)"},{"key":"1_CR13","unstructured":"Bommasani, R., et al.: On the opportunities and risks of foundation models. Stanford Center for Research on Foundation Models (2021)"},{"key":"1_CR14","unstructured":"Jurafsky, D., Martin, J.H.: Speech and Language Processing (3rd ed.). Draft Version (2021)"},{"key":"1_CR15","unstructured":"Zhang, L., et al.: Designing conversational agents for cultural contexts. In: Proceedings of the CHI Conference on Human Factors in Computing Systems (2022)"}],"container-title":["Lecture Notes on Data Engineering and Communications Technologies","Advances in Networked-Based Information Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-05892-8_1","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T13:57:43Z","timestamp":1782309463000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-05892-8_1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,1]]},"ISBN":["9783032058911","9783032058928"],"references-count":15,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-05892-8_1","relation":{},"ISSN":["2367-4512","2367-4520"],"issn-type":[{"value":"2367-4512","type":"print"},{"value":"2367-4520","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,10,1]]},"assertion":[{"value":"1 October 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"NBiS","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Network-Based Information Systems","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Taichung","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Taiwan","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"3 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"5 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"nbis2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/voyager.ce.fit.ac.jp\/conf\/nbis\/2025\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}