{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2023,2,13]],"date-time":"2023-02-13T17:53:07Z","timestamp":1676310787304},"reference-count":37,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2011,5,12]],"date-time":"2011-05-12T00:00:00Z","timestamp":1305158400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["K\u00fcnstl Intell"],"published-print":{"date-parts":[[2011,8]]},"DOI":"10.1007\/s13218-011-0108-9","type":"journal-article","created":{"date-parts":[[2011,5,11]],"date-time":"2011-05-11T15:38:09Z","timestamp":1305128289000},"page":"225-234","source":"Crossref","is-referenced-by-count":2,"title":["Computational Assessment of Interest in Speech\u2014Facing the Real-Life Challenge"],"prefix":"10.1007","volume":"25","author":[{"given":"Martin","family":"W\u00f6llmer","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Felix","family":"Weninger","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Florian","family":"Eyben","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bj\u00f6rn","family":"Schuller","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2011,5,12]]},"reference":[{"key":"108_CR1","doi-asserted-by":"crossref","first-page":"2350","DOI":"10.21437\/Interspeech.2010-643","volume-title":"Proc of interspeech","author":"L Devillers","year":"2010","unstructured":"Devillers L, Vaudable C, Chastagnol C (2010) Real-life emotion-related states detection in call centers: a cross-corpora study. In: Proc of interspeech, Makuhari, Japan, pp\u00a02350\u20132353"},{"key":"108_CR2","first-page":"1459","volume-title":"Proc of ACM multimedia","author":"F Eyben","year":"2010","unstructured":"Eyben F, W\u00f6llmer M, Schuller B (2010) openSMILE\u2014the Munich versatile and fast open-source audio feature extractor. In: Proc of ACM multimedia, Firenze, Italy, pp\u00a01459\u20131462"},{"key":"108_CR3","first-page":"489","volume-title":"Proc of ICASSP","author":"D Gatica-Perez","year":"2005","unstructured":"Gatica-Perez D, McCowan I, Zhang D, Bengio S (2005) Detecting group interest-level in meetings. In: Proc of ICASSP, Philadelphia, USA, pp\u00a0489\u2013492"},{"key":"108_CR4","first-page":"1","volume":"20","author":"A Graves","year":"2008","unstructured":"Graves A, Fernandez S, Liwicki M, Bunke H, Schmidhuber J (2008) Unconstrained online handwriting recognition with recurrent neural networks. Adv Neural Inf Process Syst 20:1\u20138","journal-title":"Adv Neural Inf Process Syst"},{"key":"108_CR5","first-page":"602","volume-title":"Proc of ICANN","author":"A Graves","year":"2005","unstructured":"Graves A, Fernandez S, Schmidhuber J (2005) Bidirectional LSTM networks for improved phoneme classification and recognition. In: Proc of ICANN, Warsaw, Poland, pp\u00a0602\u2013610"},{"key":"108_CR6","first-page":"865","volume-title":"Proc of ICME","author":"M Grimm","year":"2008","unstructured":"Grimm M, Kroschel K, Narayanan S (2008) The Vera am Mittag german audio-visual emotional speech database. In: Proc of ICME, Hannover, Germany, pp\u00a0865\u2013868"},{"key":"108_CR7","unstructured":"Hall MA (1999) Correlation-based feature selection for machine learning. PhD thesis, University of Waikato"},{"key":"108_CR8","doi-asserted-by":"crossref","first-page":"2354","DOI":"10.21437\/Interspeech.2010-644","volume-title":"Proc of interspeech","author":"A Hassan","year":"2010","unstructured":"Hassan A, Damper RI (2010) Multi-class and hierarchical SVMs for emotion recognition. In: Proc of interspeech, Makuhari, Japan, pp\u00a02354\u20132357"},{"key":"108_CR9","first-page":"1","volume-title":"A field guide to dynamical recurrent neural networks","author":"S Hochreiter","year":"2001","unstructured":"Hochreiter S, Bengio Y, Frasconi P, Schmidhuber J (2001) Gradient flow in recurrent nets: the difficulty of learning long-term dependencies. In: Kremer SC, Kolen JF (eds) A field guide to dynamical recurrent neural networks. IEEE Press, New York, pp 1\u201315"},{"issue":"8","key":"108_CR10","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter S, Schmidhuber J (1997) Long short-term memory. Neural Comput 9(8):1735\u20131780","journal-title":"Neural Comput"},{"key":"108_CR11","doi-asserted-by":"crossref","first-page":"2802","DOI":"10.21437\/Interspeech.2010-741","volume-title":"Proc of interspeech","author":"JH Jeon","year":"2010","unstructured":"Jeon JH, Xia R, Liu Y (2010) Level of interest sensing in spoken dialog using multi-level fusion of acoustic and lexical evidence. In: Proc of interspeech, Makuhari, Japan, pp\u00a02802\u20132805"},{"key":"108_CR12","first-page":"137","volume-title":"Proc of ECML","author":"T Joachims","year":"1998","unstructured":"Joachims T (1998) Text categorization with support vector machines: learning with many relevant features. In: Proc of ECML, Chemnitz, Germany, pp\u00a0137\u2013142"},{"key":"108_CR13","first-page":"243","volume-title":"Proc of ASRU","author":"L Kennedy","year":"2003","unstructured":"Kennedy L, Ellis D (2003) Pitch-based emphasis detection for characterization of meeting recordings. In: Proc of ASRU, Virgin Islands, pp\u00a0243\u2013248"},{"key":"108_CR14","first-page":"49","volume-title":"Proc of workshop on CVPR for HCI","author":"S Mota","year":"2003","unstructured":"Mota S, Picard R (2003) Automated posture analysis for detecting learner\u2019s interest level. In: Proc of workshop on CVPR for HCI, Madison, pp 49\u201355"},{"key":"108_CR15","first-page":"147","volume-title":"Proc of ICCE","author":"P Nguyen","year":"2010","unstructured":"Nguyen P, Tran D, Huang X, Sharma D (2010) Automatic classification of speaker characteristics. In: Proc of ICCE, pp 147\u2013152"},{"key":"108_CR16","volume-title":"Proc of IEEE international conference on computer vision, workshop on modeling people and human interaction (ICCV-PHI)","author":"A Pentland","year":"2005","unstructured":"Pentland A, Madan A (2005) Perception of social interest. In: Proc of IEEE international conference on computer vision, workshop on modeling people and human interaction (ICCV-PHI), Beijing, China"},{"key":"108_CR17","series-title":"Lecture notes in computer science","doi-asserted-by":"crossref","first-page":"151","DOI":"10.1007\/978-3-642-14715-9_15","volume-title":"Human behavior understanding","author":"T Pfister","year":"2010","unstructured":"Pfister T, Robinson P (2010) Speech emotion classification and public speaking skill assessment. In: Salah A, Gevers T, Sebe N, Vinciarelli A (eds) Human behavior understanding. Lecture notes in computer science, vol\u00a06219. Springer, Berlin, pp 151\u2013162"},{"key":"108_CR18","doi-asserted-by":"crossref","first-page":"586","DOI":"10.1109\/ICNN.1993.298623","volume-title":"Proc of IEEE international conference on neural networks","author":"M Riedmiller","year":"1993","unstructured":"Riedmiller M, Braun H (1993) A direct adaptive method for faster backpropagation learning: the RPROP algorithm. In: Proc of IEEE international conference on neural networks, pp 586\u2013591"},{"key":"108_CR19","first-page":"1","volume-title":"Proc of 4th intern workshop on human-computer conversation","author":"M Schr\u00f6der","year":"2008","unstructured":"Schr\u00f6der M, Cowie R, Heylen D, Pantic M, Pelachaud C, Schuller B (2008) Towards responsive sensitive artificial listeners. In: Proc of 4th intern workshop on human-computer conversation, Bellagio, Italy, pp\u00a01\u20136"},{"issue":"12","key":"108_CR20","doi-asserted-by":"crossref","first-page":"1760","DOI":"10.1016\/j.imavis.2009.02.013","volume":"27","author":"B Schuller","year":"2009","unstructured":"Schuller B, M\u00fcller R, Eyben F, Gast J, H\u00f6rnler B, W\u00f6llmer M, Rigoll G, H\u00f6thker A, Konosu H (2009) Being bored? Recognising natural interest by extensive audiovisual integration for real-life application. Image Vis Comput J 27(12):1760\u20131774. Special issue on visual and multimodal analysis of human spontaneous behavior","journal-title":"Image Vis Comput J"},{"key":"108_CR21","first-page":"1818","volume-title":"Proc of interspeech","author":"B Schuller","year":"2006","unstructured":"Schuller B, Rigoll G (2006) Timing levels in segment-based speech emotion recognition. In: Proc of interspeech, Pittsburgh, USA, pp\u00a01818\u20131821"},{"key":"108_CR22","doi-asserted-by":"crossref","first-page":"1999","DOI":"10.21437\/Interspeech.2009-484","volume-title":"Proc of interspeech","author":"B Schuller","year":"2009","unstructured":"Schuller B, Rigoll G (2009) Recognising interest in conversational speech\u2014comparing bag of frames and supra-segmental features. In: Proc of interspeech, Brighton, UK, pp\u00a01999\u20132002"},{"key":"108_CR23","doi-asserted-by":"crossref","first-page":"2794","DOI":"10.21437\/Interspeech.2010-739","volume-title":"Proc of interspeech","author":"B Schuller","year":"2010","unstructured":"Schuller B, Steidl S, Batliner A, Burkhardt F, Devillers L, M\u00fcller C, Narayanan S (2010) The interspeech 2010 paralinguistic challenge. In: Proc of interspeech, Makuhari, Japan, pp\u00a02794\u20132797"},{"key":"108_CR24","first-page":"596","volume-title":"Proc of ASRU","author":"B Schuller","year":"2007","unstructured":"Schuller B, Vlasenko B, Minguez R, Rigoll G, Wendemuth A (2007) Comparing one and two-stage acoustic modeling in the recognition of emotion in speech. In: Proc of ASRU, Kyoto, Japan, pp\u00a0596\u2013600"},{"key":"108_CR25","doi-asserted-by":"crossref","first-page":"2673","DOI":"10.1109\/78.650093","volume":"45","author":"M Schuster","year":"1997","unstructured":"Schuster M, Paliwal KK (1997) Bidirectional recurrent neural networks. IEEE Trans Signal Process 45:2673\u20132681","journal-title":"IEEE Trans Signal Process"},{"key":"108_CR26","doi-asserted-by":"crossref","first-page":"1781","DOI":"10.21437\/Interspeech.2005-3","volume-title":"Proc of interspeech","author":"E Shriberg","year":"2005","unstructured":"Shriberg E (2005) Spontaneous speech: how peoply really talk and why engineers should care. In: Proc of interspeech, Lisbon, Portugal, pp\u00a01781\u20131784"},{"issue":"4","key":"108_CR27","doi-asserted-by":"crossref","first-page":"928","DOI":"10.1109\/TNN.2002.1021893","volume":"13","author":"R Stiefelhagen","year":"2002","unstructured":"Stiefelhagen R, Yang J, Waibel A (2002) Modeling focus of attention for meeting indexing based on multiple cues. IEEE Trans Neural Netw 13(4):928\u2013938","journal-title":"IEEE Trans Neural Netw"},{"key":"108_CR28","volume-title":"Proc of ICASSP","author":"F Weninger","year":"2011","unstructured":"Weninger F, Durrieu JL, Eyben F, Richard G, Schuller B (2011) Combining monoaural source separation with long short-term memory for increased robustness in vocalist gender recognition. In: Proc of ICASSP, Prague, Czech Republic"},{"key":"108_CR29","volume-title":"Data mining: practical machine learning tools and techniques","author":"IH Witten","year":"2005","unstructured":"Witten IH, Frank E (2005) Data mining: practical machine learning tools and techniques, 2nd edn. Morgan Kaufmann, San Francisco","edition":"2"},{"issue":"1-3","key":"108_CR30","doi-asserted-by":"crossref","first-page":"366","DOI":"10.1016\/j.neucom.2009.08.005","volume":"73","author":"M W\u00f6llmer","year":"2009","unstructured":"W\u00f6llmer M, Al-Hames M, Eyben F, Schuller B, Rigoll G (2009) A multidimensional dynamic time warping algorithm for efficient multimodal fusion of asynchronous data streams. Neurocomputing 73(1-3):366\u2013380","journal-title":"Neurocomputing"},{"issue":"3","key":"108_CR31","doi-asserted-by":"crossref","first-page":"180","DOI":"10.1007\/s12559-010-9041-8","volume":"2","author":"M W\u00f6llmer","year":"2010","unstructured":"W\u00f6llmer M, Eyben F, Graves A, Schuller B, Rigoll G (2010) Bidirectional LSTM networks for context-sensitive keyword detection in a cognitive virtual agent framework. Cognit Comput 2(3):180\u2013190","journal-title":"Cognit Comput"},{"key":"108_CR32","doi-asserted-by":"crossref","first-page":"597","DOI":"10.21437\/Interspeech.2008-192","volume-title":"Proc of interspeech","author":"M W\u00f6llmer","year":"2008","unstructured":"W\u00f6llmer M, Eyben F, Reiter S, Schuller B, Cox C, Douglas-Cowie E, Cowie R (2008) Abandoning emotion classes\u2014towards continuous emotion recognition with modelling of long-range dependencies. In: Proc of interspeech, Brisbane, Australia, pp\u00a0597\u2013600"},{"key":"108_CR33","doi-asserted-by":"crossref","first-page":"1946","DOI":"10.21437\/Interspeech.2010-97","volume-title":"Proc of interspeech","author":"M W\u00f6llmer","year":"2010","unstructured":"W\u00f6llmer M, Eyben F, Schuller B, Rigoll G (2010) Recognition of spontaneous conversational speech using long short-term memory phoneme predictions. In: Proc of interspeech, Makuhari, Japan, pp\u00a01946\u20131949"},{"key":"108_CR34","volume-title":"Proc of ICASSP","author":"M W\u00f6llmer","year":"2011","unstructured":"W\u00f6llmer M, Eyben F, Schuller B, Rigoll G (2011) A multi-stream ASR framework for BLSTM modeling of conversational speech. In: Proc of ICASSP, Prague, Czech Republic"},{"key":"108_CR35","doi-asserted-by":"crossref","first-page":"2362","DOI":"10.21437\/Interspeech.2010-646","volume-title":"Proc of interspeech","author":"M W\u00f6llmer","year":"2010","unstructured":"W\u00f6llmer M, Metallinou A, Eyben F, Schuller B, Narayanan S (2010) Context-sensitive multimodal emotion recognition from speech and facial expression using bidirectional lstm modeling. In: Proc of interspeech, Makuhari, Japan, pp 2362\u20132365"},{"issue":"5","key":"108_CR36","doi-asserted-by":"crossref","first-page":"867","DOI":"10.1109\/JSTSP.2010.2057200","volume":"4","author":"M W\u00f6llmer","year":"2010","unstructured":"W\u00f6llmer M, Schuller B, Eyben F, Rigoll G (2010) Combining long short-term memory and dynamic Bayesian networks for incremental emotion-sensitive artificial listening. IEEE J Sel Top Signal Process 4(5):867\u2013881","journal-title":"IEEE J Sel Top Signal Process"},{"issue":"1","key":"108_CR37","doi-asserted-by":"crossref","first-page":"39","DOI":"10.1109\/TPAMI.2008.52","volume":"31","author":"Z Zeng","year":"2009","unstructured":"Zeng Z, Pantic M, Rosiman GI, Huang TS (2009) A survey of affect recognition methods: audio, visual, and spontaneous expressions. IEEE Trans Pattern Anal Mach Intell 31(1):39\u201358","journal-title":"IEEE Trans Pattern Anal Mach Intell"}],"container-title":["KI - K\u00fcnstliche Intelligenz"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s13218-011-0108-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s13218-011-0108-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s13218-011-0108-9","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,11,23]],"date-time":"2021-11-23T03:27:22Z","timestamp":1637638042000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s13218-011-0108-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2011,5,12]]},"references-count":37,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2011,8]]}},"alternative-id":["108"],"URL":"https:\/\/doi.org\/10.1007\/s13218-011-0108-9","relation":{},"ISSN":["0933-1875","1610-1987"],"issn-type":[{"value":"0933-1875","type":"print"},{"value":"1610-1987","type":"electronic"}],"subject":[],"published":{"date-parts":[[2011,5,12]]}}}