{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,31]],"date-time":"2025-10-31T07:31:49Z","timestamp":1761895909041},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"22","license":[{"start":{"date-parts":[[2014,7,16]],"date-time":"2014-07-16T00:00:00Z","timestamp":1405468800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2015,11]]},"DOI":"10.1007\/s11042-014-2161-5","type":"journal-article","created":{"date-parts":[[2014,7,15]],"date-time":"2014-07-15T05:50:21Z","timestamp":1405403421000},"page":"10025-10051","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":8,"title":["User behavior fusion in dialog management with multi-modal history cues"],"prefix":"10.1007","volume":"74","author":[{"given":"Minghao","family":"Yang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianhua","family":"Tao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Linlin","family":"Chao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hao","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dawei","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hao","family":"Che","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tingli","family":"Gao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bin","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2014,7,16]]},"reference":[{"key":"2161_CR1","unstructured":"Ananova. http:\/\/en.wikipedia.org\/wiki\/Ananova . Accessed 18 Jan 2014; Available from: http:\/\/en.wikipedia.org\/wiki\/Ananova"},{"key":"2161_CR2","doi-asserted-by":"crossref","unstructured":"Baltru\u0161aitis T, Ramirez GA, Morency L-P (2011) Modeling latent discriminative dynamic of multi-dimensional affective signals. Affect Comput Intell Interact, p 396\u2013406. Springer, Berlin","DOI":"10.1007\/978-3-642-24571-8_51"},{"key":"2161_CR3","doi-asserted-by":"crossref","unstructured":"Bell L, Gustafson J (2000) Positive and negative user feedback in a spoken dialogue Corpus. In: INTERSPEECH. p 589\u2013592","DOI":"10.21437\/ICSLP.2000-146"},{"key":"2161_CR4","first-page":"378","volume":"3975","author":"N Bianchi-Berthouze","year":"2011","unstructured":"Bianchi-Berthouze N, Meng H (2011) Naturalistic affective expression classification by a multi-stage approach based on hidden Markov models. Affect Comput Intell Interact 3975:378\u2013387","journal-title":"Affect Comput Intell Interact"},{"key":"2161_CR5","unstructured":"Bohus D, Rudnicky A (2005) Sorry, I didn\u2019t catch that! - an investigation of non-understanding errors and recovery strategies. In: Proceedings of SIGdial. Lisbon, Portugal"},{"key":"2161_CR6","doi-asserted-by":"crossref","unstructured":"Bousmalis K, Zafeiriou S, Morency L-P, Pantic M, Ghahramani Z (2013) Variational hidden conditional random fields with coupled Dirichlet process mixtures. In: European conference on machine learning and principles and practice of knowledge discovery in databases","DOI":"10.1007\/978-3-642-40991-2_34"},{"key":"2161_CR7","unstructured":"Brustoloni JC (1991) Autonomous agents: characterization and requirements"},{"key":"2161_CR8","unstructured":"Cerekovic TPA, Igor P (2009) RealActor: character animation and multimodal behavior realization system. Intelligent virtual agents. p 486\u2013487."},{"key":"2161_CR9","doi-asserted-by":"crossref","first-page":"1","DOI":"10.5772\/54002","volume":"10","author":"S Dobrisek","year":"2013","unstructured":"Dobrisek S, Gajsek R, Mihelic F, Pavesic N, Struc V (2013) Towards efficient multi-modal emotion recognition. Int J Adv Robot Syst 10:1\u201310","journal-title":"Int J Adv Robot Syst"},{"issue":"3","key":"2161_CR10","doi-asserted-by":"crossref","first-page":"235","DOI":"10.1080\/09588220701489507","volume":"20","author":"O Engwall","year":"2007","unstructured":"Engwall O, Balter O (2007) Pronunciation feedback from real and virtual language teachers. J Comput Assist Lang Learn 20(3):235\u2013262","journal-title":"J Comput Assist Lang Learn"},{"key":"2161_CR11","doi-asserted-by":"crossref","unstructured":"Goddeau HMD, Poliforni J, Seneff S, Busayapongchait S (1996) A form-based dialogue management for spoken language applications. In: International conference on spoken language processing. Pittsburgh, PA. p 701\u2013704","DOI":"10.1109\/ICSLP.1996.607458"},{"key":"2161_CR12","unstructured":"GoogleAPI. www.google.com\/intl\/en\/chrome\/demos\/speech.heml . Available from: www.google.com\/intl\/en\/chrome\/demos\/speech.heml"},{"key":"2161_CR13","unstructured":"Heloir A, Kipp M, Gebhard P, Schroeder M (2010) Realizing Multimodal Behavior: Closing the gap between behavior planning and embodied agent presentation. In: Proceedings of the 10th international conference on intelligent virtual agents. Springer"},{"issue":"10","key":"2161_CR14","doi-asserted-by":"crossref","first-page":"1024","DOI":"10.1016\/j.specom.2009.05.006","volume":"51","author":"A Hjalmarsson","year":"2009","unstructured":"Hjalmarsson A, Wik P (2009) Embodied conversational agents in computer assisted language learning. Speech Comm 51(10):1024\u20131037","journal-title":"Speech Comm"},{"issue":"2","key":"2161_CR15","doi-asserted-by":"crossref","first-page":"137","DOI":"10.1023\/B:VISI.0000013087.49260.fb","volume":"57","author":"MJ Jones","year":"2004","unstructured":"Jones MJ, Viola PA (2004) Robust real-time face detection. Int J Comput Vis 57(2):137\u2013154","journal-title":"Int J Comput Vis"},{"key":"2161_CR16","first-page":"153","volume":"31","author":"M Kaiser","year":"2012","unstructured":"Kaiser M, Willmer M, Eyben F, Schuller B (2012) LSTM-Modeling of continuous emotions in an audiovisual affect recognition framework. Image Vis Comput 31:153\u2013163","journal-title":"Image Vis Comput"},{"key":"2161_CR17","unstructured":"Kang Y, Tao J (2005) Features importance analysis for emotion speech classification. In: International conference on affective computing and intelligence interaction -ACII 2005. p 449\u2013457"},{"key":"2161_CR18","unstructured":"kth. http:\/\/www.speech.kth.se\/multimodal\/ . Accessed 18 Jan 2014; Available from: http:\/\/www.speech.kth.se\/multimodal\/"},{"issue":"1","key":"2161_CR19","doi-asserted-by":"crossref","first-page":"1","DOI":"10.5626\/JCSE.2010.4.1.001","volume":"4","author":"C Lee","year":"2010","unstructured":"Lee C, Jung S, Kim K, Lee D, Lee GG (2010) Recent approaches to dialog management for spoken dialog systems. J Comput Sci Eng 4(1):1\u201322","journal-title":"J Comput Sci Eng"},{"issue":"1","key":"2161_CR20","doi-asserted-by":"crossref","first-page":"11","DOI":"10.1109\/89.817450","volume":"8","author":"RPE Levin","year":"2000","unstructured":"Levin RPE, Eckert W (2000) A stochastic model of human-machine interaction for learning dialog strategies. IEEE Trans Speech Audio Process 8(1):11\u201323","journal-title":"IEEE Trans Speech Audio Process"},{"key":"2161_CR21","doi-asserted-by":"crossref","unstructured":"Litman DJ, Tetreault JR (2006) Comparing the utility of state features in spoken dialogue using reinforcement learning. In: Conference: North American Chapter of the association for computational linguistics - NAACL. New York City","DOI":"10.3115\/1220835.1220870"},{"key":"2161_CR22","unstructured":"MapAPIBaidu. http:\/\/developer.baidu.com\/map\/webservice.htm . Available from: http:\/\/developer.baidu.com\/map\/webservice.htm"},{"key":"2161_CR23","doi-asserted-by":"crossref","unstructured":"McKeown G, Valstar MF, Cowie R, Pantic M (2010) The SEMAINE Corpus of emotionally coloured character interactions. In: Proc IEEE Int\u2019l Conf Multimedia and Expo. p 1079\u20131084","DOI":"10.1109\/ICME.2010.5583006"},{"key":"2161_CR24","unstructured":"mmdagent. http:\/\/www.mmdagent.jp\/ . Accessed 18 Jan 2014; Available from: http:\/\/www.mmdagent.jp\/"},{"key":"2161_CR25","unstructured":"nlprFace. http:\/\/www.cripac.ia.ac.cn\/Databases\/databases.html and http:\/\/www.idealtest.org\/dbDetailForUser.do?id=9 . Accessed 18 Jan 2014; Available from: http:\/\/www.cripac.ia.ac.cn\/Databases\/databases.html and http:\/\/www.idealtest.org\/dbDetailForUser.do?id=9"},{"issue":"2","key":"2161_CR26","doi-asserted-by":"crossref","first-page":"589","DOI":"10.1109\/TSA.2005.855836","volume":"14","author":"TDO Pietquin","year":"2006","unstructured":"Pietquin TDO (2006) A probabilistic framework for dialog simulation and optimal strategy. IEEE Trans Audio Speech Lang Process 14(2):589\u2013599","journal-title":"IEEE Trans Audio Speech Lang Process"},{"key":"2161_CR27","unstructured":"Rebillat M, Courgeon M, Katz B, Clavel C, Martin J-C (2010) Life-sized audiovisual spatial social scenes with multiple characters: MARC SMART-I2. In: 5th meeting of the French association for virtual reality"},{"issue":"2","key":"2161_CR28","doi-asserted-by":"crossref","first-page":"97","DOI":"10.1017\/S0269888906000944","volume":"21","author":"KWJ Schatzmann","year":"2006","unstructured":"Schatzmann KWJ, Stuttle M, Young S (2006) A survey of statistical user simulation techniques for reinforcement-learning of dialogue management strategies. Knowl Eng Rev 21(2):97\u2013126","journal-title":"Knowl Eng Rev"},{"key":"2161_CR29","doi-asserted-by":"crossref","unstructured":"Schels M, Glodek M, Meudt S, Scherer S, Schmidt M, Layher G, Tschechne S, Brosch T, Hrabal D, Walter S, Traue HC, Palm G, Schwenker F, Campbell MR (2013) Multi-modal classifier-fusion for the recognition of emotions, chapter in converbal synchrony in human-machine interaction. CRC Press, Boca Raton, FL 33487, USA","DOI":"10.1201\/b15477-5"},{"key":"2161_CR30","volume-title":"Handbook of cognition and emotion","author":"KR Scherer","year":"1999","unstructured":"Scherer KR (1999) Appraisal theory. In: Dalgleish T, Power M (eds) Handbook of cognition and emotion. Wiley, Chichester"},{"key":"2161_CR31","doi-asserted-by":"crossref","unstructured":"Schwarzlery SMS, Schenk J, Wallhoff F, Rigoll G (2009) Using graphical models for mixed-initiative dialog management systems with realtime Policies. In: Conference: annual conference of the International Speech Communication Association - INTERSPEECH. p 260\u2013263","DOI":"10.21437\/Interspeech.2009-90"},{"issue":"10","key":"2161_CR32","doi-asserted-by":"crossref","first-page":"897","DOI":"10.1109\/LSP.2009.2026457","volume":"16","author":"S Shan","year":"2009","unstructured":"Shan S, Niu Z, Chen X (2009) Facial shape localization using probability gradient hints. IEEE Signal Process Lett 16(10):897\u2013900","journal-title":"IEEE Signal Process Lett"},{"key":"2161_CR33","unstructured":"SPTK. http:\/\/sp-tk.sourceforge.jp . Accessed 18 Jan 2014; Available from: http:\/\/sp-tk.sourceforge.jp"},{"key":"2161_CR34","unstructured":"Steedman M, Badler N, Achorn B, Bechet T, Douville B, Prevost S, Cassell J, Pelachaud C, Stone M (1994) Animated conversation: Rule-based generation of facial expression gesture and spoken intonation for multiple conversation agents. In: Proceedings of SIGGRAPH. p 73\u201380"},{"issue":"1","key":"2161_CR35","first-page":"1","volume":"11","author":"J Tao","year":"2011","unstructured":"Tao J, Pan S, Yang M, Li Y, Mu K, Che J (2011) Utterance independent bimodal emotion recognition in spontaneous communication. EURASIP J Adv Signal Process 11(1):1\u201311","journal-title":"EURASIP J Adv Signal Process"},{"issue":"1\u20132","key":"2161_CR36","first-page":"61","volume":"5","author":"J Tao","year":"2012","unstructured":"Tao J, Yang M, Mu K, Li Y, Che J (2012) A multimodal approach of generating 3D human-like talking agent. J Multimodal User Interfaces 5(1\u20132):61\u201368","journal-title":"J Multimodal User Interfaces"},{"key":"2161_CR37","unstructured":"Tao J, Yang M, Chao L (2013) Combining emotional history through multimodal fusion methods. In: Asia Pacific Signal and Information Processing Association (APSIPA 2013). Kaohsiung, Taiwan, China"},{"key":"2161_CR38","doi-asserted-by":"crossref","unstructured":"Tschechne S, Glodek M, Layher G, Schels M, Brosch T, Scherer S, Schwenker F (2011) Multiple classifier systems for the classification of audio-visual emotion states. Affect Comput Intell Interact, 378\u2013387. Springer, Berlin","DOI":"10.1007\/978-3-642-24571-8_47"},{"issue":"4","key":"2161_CR39","first-page":"271","volume":"3","author":"D Reidsma Van","year":"2010","unstructured":"Van Reidsma D, Welbergen H, Ruttkay ZM, Zwiers Elckerlyc J (2010) A BML Realizer for continuous, multimodal interaction with a Virtual Human. J Multimodal User Interfaces 3(4):271\u2013284","journal-title":"J Multimodal User Interfaces"},{"key":"2161_CR40","unstructured":"Williams JD (2003) A probabilistic model of human\/computer dialogue with application to a partially observable Markov decision process"},{"key":"2161_CR41","unstructured":"Williams JD, Poupart P, Young S (2005) Partially observable Markov decision processes with continuous observations for dialogue management. In: Proceedings of the 6th SigDial workshop on discourse and dialogue. Lisbon"},{"key":"2161_CR42","unstructured":"Xin L, Huang L, Zhao L, Tao J (2007) Combining audio and video by dominance in bimodal emotion recognition. In: International conference on affective computing and intelligence interaction - ACII. p 729\u2013730"},{"key":"2161_CR43","doi-asserted-by":"crossref","unstructured":"Young S (2006) Using POMDPs for dialog management. In: Conference: IEEE workshop on spoken language technology - SLT","DOI":"10.1109\/SLT.2006.326785"},{"issue":"1","key":"2161_CR44","doi-asserted-by":"crossref","first-page":"39","DOI":"10.1109\/TPAMI.2008.52","volume":"31","author":"Z Zeng","year":"2009","unstructured":"Zeng Z, Pantic M, Roisman GI, Huang TS (2009) A survey of affect recognition methods: audio, visual, and spontaneous expressions. IEEE Trans Pattern Anal Mach Intell 31(1):39\u201358","journal-title":"IEEE Trans Pattern Anal Mach Intell"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-014-2161-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11042-014-2161-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-014-2161-5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,7,15]],"date-time":"2023-07-15T07:24:39Z","timestamp":1689405879000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11042-014-2161-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014,7,16]]},"references-count":44,"journal-issue":{"issue":"22","published-print":{"date-parts":[[2015,11]]}},"alternative-id":["2161"],"URL":"https:\/\/doi.org\/10.1007\/s11042-014-2161-5","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"value":"1380-7501","type":"print"},{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2014,7,16]]}}}