{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,12]],"date-time":"2025-11-12T13:58:25Z","timestamp":1762955905827,"version":"3.41.0"},"publisher-location":"Cham","reference-count":25,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319912493"},{"type":"electronic","value":"9783319912509"}],"license":[{"start":{"date-parts":[[2018,1,1]],"date-time":"2018-01-01T00:00:00Z","timestamp":1514764800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2018,1,1]],"date-time":"2018-01-01T00:00:00Z","timestamp":1514764800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2018]]},"DOI":"10.1007\/978-3-319-91250-9_13","type":"book-chapter","created":{"date-parts":[[2018,5,31]],"date-time":"2018-05-31T11:11:55Z","timestamp":1527765115000},"page":"164-175","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Comparing Cascaded LSTM Architectures for Generating Head Motion from Speech in Task-Oriented Dialogs"],"prefix":"10.1007","author":[{"given":"Duc-Canh","family":"Nguyen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"G\u00e9rard","family":"Bailly","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fr\u00e9d\u00e9ric","family":"Elisei","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2018,6,1]]},"reference":[{"key":"13_CR1","doi-asserted-by":"crossref","unstructured":"Alahi, A., Goel, K., Ramanathan, V., Robicquet, A., Fei-Fei, L., Savarese, S.: Social LSTM: Human trajectory prediction in crowded spaces. In: IEEE Conference on Computer Vision and Pattern Recognition (CPVR), pp. 961\u2013971 (2016)","DOI":"10.1109\/CVPR.2016.110"},{"key":"13_CR2","doi-asserted-by":"crossref","unstructured":"Ben Youssef, A., Shimodaira, H., Braude, D.A.: Articulatory features for speech-driven head motion synthesis. In: Interspeech, pp. 2758\u20132762 (2013)","DOI":"10.21437\/Interspeech.2013-632"},{"key":"13_CR3","unstructured":"Boersma, P., Weenik, D.: PRAAT: a system for doing phonetics by computer. Report of the Institute of Phonetic Sciences of the University of Amsterdam. University of Amsterdam, Amsterdam (1996)"},{"issue":"12","key":"13_CR4","doi-asserted-by":"publisher","first-page":"e83068","DOI":"10.1371\/journal.pone.0083068","volume":"8","author":"WO Brimijoin","year":"2013","unstructured":"Brimijoin, W.O., Boyd, A.W., Akeroyd, M.A.: The contribution of head movement to the externalization and internalization of sounds. PloS One 8(12), e83068 (2013)","journal-title":"PloS One"},{"issue":"3","key":"13_CR5","doi-asserted-by":"publisher","first-page":"1075","DOI":"10.1109\/TASL.2006.885910","volume":"15","author":"C Busso","year":"2007","unstructured":"Busso, C., Deng, Z., Grimm, M., Neumann, U., Narayanan, S.: Rigid head motion in expressive speech animation: analysis and synthesis. IEEE Trans. Audio Speech Lang. Process. 15(3), 1075\u20131086 (2007)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"13_CR6","doi-asserted-by":"crossref","unstructured":"Cassell, J., Pelachaud, C., Badler, N., Steedman, M., Achorn, B., Becket, T., Douville, B., Prevost, S., Stone, M.: Animated conversation: rule-based generation of facial expression, gesture & spoken intonation for multiple conversational agents. In: Annual Conference on Computer Graphics and Interactive Techniques, pp. 413\u2013420. ACM (1994)","DOI":"10.1145\/192161.192272"},{"key":"13_CR7","series-title":"Studies in Classification, Data Analysis, and Knowledge Organization","doi-asserted-by":"publisher","first-page":"321","DOI":"10.1007\/978-3-642-59789-3_51","volume-title":"Data Analysis, Classification, and Related Methods","author":"C Dehon","year":"2000","unstructured":"Dehon, C., Filzmoser, P., Croux, C.: Robust methods for canonical correlation analysis. In: Kiers, H.A.L., Rasson, J.P., Groenen, P.J.F., Schader, M. (eds.) Data Analysis, Classification, and Related Methods. Studies in Classification, Data Analysis, and Knowledge Organization, pp. 321\u2013326. Springer, Heidelberg (2000). https:\/\/doi.org\/10.1007\/978-3-642-59789-3_51"},{"key":"13_CR8","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"217","DOI":"10.1007\/978-3-642-40415-3_19","volume-title":"Intelligent Virtual Agents","author":"Y Ding","year":"2013","unstructured":"Ding, Y., Pelachaud, C., Arti\u00e8res, T.: Modeling multimodal behaviors from speech prosody. In: Aylett, R., Krenn, B., Pelachaud, C., Shimodaira, H. (eds.) IVA 2013. LNCS (LNAI), vol. 8108, pp. 217\u2013228. Springer, Heidelberg (2013). https:\/\/doi.org\/10.1007\/978-3-642-40415-3_19"},{"key":"13_CR9","doi-asserted-by":"crossref","unstructured":"Graf, H.P., Cosatto, E., Strom, V., Jie Huang, F.: Visual prosody: Facial movements accompanying speech. In: Automatic Face and Gesture Recognition (FG), pp. 396\u2013401. IEEE (2002)","DOI":"10.1109\/AFGR.2002.1004186"},{"issue":"3","key":"13_CR10","doi-asserted-by":"publisher","first-page":"427","DOI":"10.1152\/jn.1987.58.3.427","volume":"58","author":"D Guitton","year":"1987","unstructured":"Guitton, D., Volle, M.: Gaze control in humans: eye-head coordination during orienting movements to targets within and beyond the oculomotor range. J. Neurophysiol. 58(3), 427\u2013459 (1987)","journal-title":"J. Neurophysiol."},{"key":"13_CR11","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"198","DOI":"10.1007\/978-3-319-47665-0_18","volume-title":"Intelligent Virtual Agents","author":"K Haag","year":"2016","unstructured":"Haag, K., Shimodaira, H.: Bidirectional LSTM networks employing stacked bottleneck features for expressive speech-driven head motion synthesis. In: Traum, D., Swartout, W., Khooshabeh, P., Kopp, S., Scherer, S., Leuski, A. (eds.) IVA 2016. LNCS (LNAI), vol. 10011, pp. 198\u2013207. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-47665-0_18"},{"key":"13_CR12","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"243","DOI":"10.1007\/11821830_20","volume-title":"Intelligent Virtual Agents","author":"J Lee","year":"2006","unstructured":"Lee, J., Marsella, S.: Nonverbal behavior generator for embodied conversational agents. In: Gratch, J., Young, M., Aylett, R., Ballin, D., Olivier, P. (eds.) IVA 2006. LNCS (LNAI), vol. 4133, pp. 243\u2013255. Springer, Heidelberg (2006). https:\/\/doi.org\/10.1007\/11821830_20"},{"issue":"5","key":"13_CR13","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/1618452.1618518","volume":"28","author":"Sergey Levine","year":"2009","unstructured":"Levine, S., Theobalt, C., Koltun, V.: Real-time prosody-driven synthesis of body language. In: ACM Transactions on Graphics (TOG), vol. 28, Article no. 172. ACM (2009)","journal-title":"ACM Transactions on Graphics"},{"key":"13_CR14","doi-asserted-by":"crossref","unstructured":"Liu, C., Ishi, C.T., Ishiguro, H., Hagita, N.: Generation of nodding, head tilting and eye gazing for human-robot dialogue interaction. In: Human-Robot Interaction (HRI), pp. 285\u2013292. IEEE (2012)","DOI":"10.1145\/2157689.2157797"},{"issue":"8","key":"13_CR15","doi-asserted-by":"publisher","first-page":"2329","DOI":"10.1109\/TASL.2012.2201476","volume":"20","author":"S Mariooryad","year":"2012","unstructured":"Mariooryad, S., Busso, C.: Generating human-like behaviors using joint, speech-driven models for conversational agents. IEEE Trans. Audio Speech Lang. Process. 20(8), 2329\u20132340 (2012)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"13_CR16","doi-asserted-by":"crossref","unstructured":"May, T., Ma, N., Brown, G.J.: Robust localisation of multiple speakers exploiting head movements and multi-conditional training of binaural cues. In: Acoustics, Speech and Signal Processing (ICASSP), pp. 2679\u20132683. IEEE (2015)","DOI":"10.1109\/ICASSP.2015.7178457"},{"issue":"2016","key":"13_CR17","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1016\/j.patrec.2016.02.005","volume":"74","author":"A Mihoub","year":"2016","unstructured":"Mihoub, A., Bailly, G., Wolf, C., Elisei, F.: Graphical models for social behavior modeling in face-to face interaction. Pattern Recogn. Lett. (PRL) 74(2016), 82\u201389 (2016)","journal-title":"Pattern Recogn. Lett. (PRL)"},{"issue":"2","key":"13_CR18","doi-asserted-by":"publisher","first-page":"133","DOI":"10.1111\/j.0963-7214.2004.01502010.x","volume":"15","author":"KG Munhall","year":"2004","unstructured":"Munhall, K.G., Jones, J.A., Callan, D.E., Kuratate, T., Vatikiotis-Bateson, E.: Visual prosody and speech intelligibility: Head movement improves auditory speech perception. Psychol. Sci. 15(2), 133\u2013137 (2004)","journal-title":"Psychol. Sci."},{"key":"13_CR19","doi-asserted-by":"crossref","unstructured":"Nguyen, D.-C., Bailly, G., Elisei, F.: Conducting neuropsychological tests with a humanoid robot: design and evaluation. In: Cognitive Infocommunications (CogInfoCom), pp. 337\u2013342. IEEE (2016)","DOI":"10.1109\/CogInfoCom.2016.7804572"},{"key":"13_CR20","unstructured":"Nguyen, D.-C., Bailly, G., Elisei, F.: Learning Off-line vs. On-line models of interactive multimodal behaviors with recurrent neural networks. Pattern Recognition Letters (PRL) (accepted with minor revision)"},{"key":"13_CR21","unstructured":"Sadoughi, N., Busso, C.: Speech-driven Animation with Meaningful Behaviors (2017). arXiv preprint arXiv:1708.01640"},{"key":"13_CR22","series-title":"Text, Speech and Language Technology","doi-asserted-by":"publisher","first-page":"173","DOI":"10.1007\/978-94-017-2367-1_8","volume-title":"Multimodality in Language and Speech Systems","author":"KR Th\u00f3risson","year":"2002","unstructured":"Th\u00f3risson, K.R.: Natural turn-taking needs no manual: computational theory and model, from perception to action. In: Granstr\u00f6m, B., House, D., Karlsson, I. (eds.) Multimodality in Language and Speech Systems. Text, Speech and Language Technology, vol. 19, pp. 173\u2013207. Springer, Dordrecht (2002). https:\/\/doi.org\/10.1007\/978-94-017-2367-1_8"},{"key":"13_CR23","unstructured":"Wittenburg, P., Brugman, H., Russel, A., Klassmann, A., Sloetjes, H.: ELAN: a professional framework for multimodality research. In: International Conference on Language Resources and Evaluation (LREC) (2006)"},{"key":"13_CR24","unstructured":"Yehia, H., Kuratate, T., Vatikiotis-Bateson, E.: Facial animation and head motion driven by speech acoustics. In: 5th Seminar on Speech Production: Models and Data, pp. 265\u2013268. Kloster Seeon, Germany (2000)"},{"issue":"1431","key":"13_CR25","doi-asserted-by":"publisher","first-page":"593","DOI":"10.1098\/rstb.2002.1238","volume":"358","author":"DM Wolpert","year":"2003","unstructured":"Wolpert, D.M., Doya, K., Kawato, M.: A unifying computational framework for motor control and social interaction. Philos. Trans. R. Soc. B Biol. Sci. 358(1431), 593\u2013602 (2003)","journal-title":"Philos. Trans. R. Soc. B Biol. Sci."}],"container-title":["Lecture Notes in Computer Science","Human-Computer Interaction. Interaction Technologies"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-91250-9_13","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,4]],"date-time":"2025-07-04T22:18:49Z","timestamp":1751667529000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-319-91250-9_13"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018]]},"ISBN":["9783319912493","9783319912509"],"references-count":25,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-91250-9_13","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2018]]},"assertion":[{"value":"1 June 2018","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"HCI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Human-Computer Interaction","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Las Vegas, NV","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"USA","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2018","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"15 July 2018","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 July 2018","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"hci2018","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/2018.hci.international\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"This content has been made available to all.","name":"free","label":"Free to read"}]}}