{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T00:37:25Z","timestamp":1783557445637,"version":"3.55.0"},"publisher-location":"Cham","reference-count":36,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030687892","type":"print"},{"value":"9783030687908","type":"electronic"}],"license":[{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021]]},"DOI":"10.1007\/978-3-030-68790-8_1","type":"book-chapter","created":{"date-parts":[[2021,2,22]],"date-time":"2021-02-22T13:13:24Z","timestamp":1613999604000},"page":"5-19","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":49,"title":["Towards Robust Deep Neural Networks for Affect and Depression Recognition from Speech"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3442-0578","authenticated-orcid":false,"given":"Alice","family":"Othmani","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Daoud","family":"Kadoch","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kamil","family":"Bentounes","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Emna","family":"Rejaibi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Romain","family":"Alfred","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Abdenour","family":"Hadid","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2021,2,23]]},"reference":[{"key":"1_CR1","unstructured":"GBD 2015 Disease and Injury Incidence and Prevalence Collaborators: Global, regional, and national incidence, prevalence, and years lived with disability for 310 diseases and injuries, 1990\u20132015: a systematic analysis for the Global Burden of Disease Study 2015, Lancet, vol. 388, no. 10053, pp. 1545\u20131602 (2015)"},{"key":"1_CR2","unstructured":"The National Institute of Mental Health: Depression. https:\/\/www.nimh.nih.gov\/health\/topics\/depression\/index.shtml. Accessed 17 June 2019"},{"key":"1_CR3","doi-asserted-by":"crossref","unstructured":"Valstar, M., et al.: AVEC 2016 - depression, mood, and emotion recognition workshop and challenge. In: Proceedings of the 6th International Workshop on Audio\/visual Emotion Challenge, pp. 3\u201310. ACM (2016)","DOI":"10.1145\/2988257.2988258"},{"key":"1_CR4","doi-asserted-by":"crossref","unstructured":"Ringeval, F., et al.: AVEC 2017 - real-life depression, and affect recognition workshop and challenge. In: Proceedings of the 7th Annual Workshop on Audio\/Visual Emotion Challenge, pp. 3\u20139. ACM (2017)","DOI":"10.1145\/3133944.3133953"},{"key":"1_CR5","doi-asserted-by":"crossref","unstructured":"Jiang, H., Hu, B., Liu, Z., Wang, G., Zhang, L., Li, X., Kang, H.: Detecting depression using an ensemble logistic regression model based on multiple speech features. Comput. Math. Methods Medicine 2018 (2018)","DOI":"10.1155\/2018\/6508319"},{"key":"1_CR6","doi-asserted-by":"crossref","unstructured":"Alghowinem, S., et al.: A comparative study of different classifiers for detecting depression from spontaneous speech. In: 2013 IEEE International Conference on Acoustics, Speech and Signal Processing, pp. 8022\u20138026 (2013)","DOI":"10.1109\/ICASSP.2013.6639227"},{"key":"1_CR7","doi-asserted-by":"crossref","unstructured":"Valstar, M., et al.: AVEC 2013: the continuous audio\/visual emotion and depression recognition challenge. In: Proceedings of the 3rd ACM International Workshop on Audio\/Visual Emotion Challenge, pp. 3\u201310 (2013)","DOI":"10.1145\/2512530.2512533"},{"key":"1_CR8","doi-asserted-by":"crossref","unstructured":"Yang, L., Sahli, H., Xia, X., Pei, E., Oveneke, M.C., Jiang, D.: Hybrid depression classification and estimation from audio video and text information. In: Proceedings of the 7th Annual Workshop on Audio\/Visual Emotion Challenge, pp. 45\u201351. ACM (2017)","DOI":"10.1145\/3133944.3133950"},{"key":"1_CR9","doi-asserted-by":"crossref","unstructured":"Cummins, N., Epps, J., Breakspear M., Goecke, R.: An investigation of depressed speech detection: features and normalization. In: Twelfth Annual Conference of the International Speech Communication Association (2011)","DOI":"10.21437\/Interspeech.2011-750"},{"key":"1_CR10","doi-asserted-by":"crossref","unstructured":"Lopez-Otero, P., Dacia-Fernandez, L., Garcia-Mateo, C.: A study of acoustic features for depression detection. In: 2nd International Workshop on Biometrics and Forensics, pp. 1\u20136. IEEE (2014)","DOI":"10.1109\/IWBF.2014.6914245"},{"key":"1_CR11","doi-asserted-by":"crossref","unstructured":"Ringeval, F., et al.: Av+EC 2015 - the first affect recognition challenge bridging across audio, video, and physiological data. In: Proceedings of the 5th International Workshop on Audio\/Visual Emotion Challenge, pp. 3\u20138. ACM (2015)","DOI":"10.1145\/2808196.2811642"},{"key":"1_CR12","doi-asserted-by":"crossref","unstructured":"He, L., Jiang, D., Yang, L., Pei, E., Wu, P., Sahli, H.: Multimodal affective dimension prediction using deep bidirectional long short-term memory recurrent neural networks. In: Proceedings of the 5th International Workshop on Audio\/Visual Emotion Challenge, pp. 73\u201380. ACM (2015)","DOI":"10.1145\/2808196.2811641"},{"key":"1_CR13","doi-asserted-by":"crossref","unstructured":"Ringeval, F., et al.: AVEC 2018 workshop and challenge: bipolar disorder and cross-cultural affect recognition. In: Proceedings of the 2018 on Audio\/Visual Emotion Challenge and Workshop, pp. 3\u201313. ACM (2018)","DOI":"10.1145\/3266302.3266316"},{"key":"1_CR14","doi-asserted-by":"crossref","unstructured":"Dhall, A., Ramana Murthy, O.V., Goecke, R., Joshi, J., Gedeon, T.: Video and image based emotion recognition challenges in the wild: EmotiW 2015. In: Proceedings of the 2015 ACM on International Conference on Multimodal Interaction, pp. 423\u2013426 (2015)","DOI":"10.1145\/2818346.2829994"},{"key":"1_CR15","unstructured":"Haq, S., Jackson, P.J., Edge, J.: Speaker-dependent audio-visual emotion recognition. In: AVSP, pp. 53\u201358 (2009)"},{"issue":"3","key":"1_CR16","doi-asserted-by":"publisher","first-page":"574","DOI":"10.1109\/TBME.2010.2091640","volume":"58","author":"LSA Low","year":"2010","unstructured":"Low, L.S.A., Maddage, N.C., Lech, M., Sheeber, L.B., Allen, N.B.: Detection of clinical depression in adolescents\u2019 speech during family interactions. IEEE Trans. Biomed. Eng. 58(3), 574\u2013586 (2010)","journal-title":"IEEE Trans. Biomed. Eng."},{"key":"1_CR17","doi-asserted-by":"crossref","unstructured":"Valstar, M., Schuller, B.W., Krajewski, J., Cowie, R., Pantic, M.: AVEC 2014: the 4th international audio\/visual emotion challenge and workshop. In: Proceedings of the 22nd ACM International Conference on Multimedia, pp. 1243\u20131244 (2014)","DOI":"10.1145\/2647868.2647869"},{"key":"1_CR18","doi-asserted-by":"crossref","unstructured":"Meng, H., Huang, D., Wang, H., Yang, H., Ai-Shuraifi, M., Wang, Y.: Depression recognition based on dynamic facial and vocal expression features using partial least square regression. In: Proceedings of the 3rd ACM International Workshop on Audio\/Visual Emotion Challenge, pp. 21\u201330 (2013)","DOI":"10.1145\/2512530.2512532"},{"key":"1_CR19","doi-asserted-by":"crossref","unstructured":"Trigeorgis, G., et al.: Adieu features? End-to-end speech emotion recognition using a deep convolutional recurrent network. In: 2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5200\u20135204 (2016)","DOI":"10.1109\/ICASSP.2016.7472669"},{"key":"1_CR20","doi-asserted-by":"publisher","first-page":"22","DOI":"10.1016\/j.patrec.2014.11.007","volume":"66","author":"F Ringeval","year":"2015","unstructured":"Ringeval, F., et al.: Prediction of asynchronous dimensional emotion ratings from audiovisual and physiological data. Pattern Recogn. Lett. 66, 22\u201330 (2015)","journal-title":"Pattern Recogn. Lett."},{"key":"1_CR21","doi-asserted-by":"crossref","unstructured":"Ringeval, F., Schuller, B., Valstar, M., Cowie, R., Pantic, M.: AVEC 2015: the 5th international audio\/visual emotion challenge and workshop. In: Proceedings of the 23rd ACM International Conference on Multimedia, pp. 1335\u20131336 (2015)","DOI":"10.1145\/2733373.2806408"},{"issue":"8","key":"1_CR22","doi-asserted-by":"publisher","first-page":"1301","DOI":"10.1109\/JSTSP.2017.2764438","volume":"11","author":"P Tzirakis","year":"2017","unstructured":"Tzirakis, P., Trigeorgis, G., Nicolaou, M.A., Schuller, B.W., Zafeiriou, S.: End-to-end multimodal emotion recognition using deep neural networks. IEEE J. Sel. Topics Signal Process. 11(8), 1301\u20131309 (2017)","journal-title":"IEEE J. Sel. Topics Signal Process."},{"key":"1_CR23","doi-asserted-by":"crossref","unstructured":"Al Hanai, T., Ghassemi, M.M., Glass, J.R.: Detecting depression with audio\/text sequence modeling of interviews. In: Interspeech, pp. 1716\u20131720 (2018)","DOI":"10.21437\/Interspeech.2018-2522"},{"key":"1_CR24","unstructured":"Dham, S., Sharma, A., Dhall, A.: Depression scale recognition from audio, visual and text analysis. arXiv preprint arXiv:1709.05865"},{"issue":"2","key":"1_CR25","first-page":"81","volume":"2","author":"A Salekin","year":"2018","unstructured":"Salekin, A., Eberle, J.W., Glenn, J.J., Teachman, B.A., Stankovic, J.A.: A weakly supervised learning framework for detecting social anxiety and depression. Proc. ACM Interact. Mobile Wearable Ubiquit. Technol. 2(2), 81 (2018)","journal-title":"Proc. ACM Interact. Mobile Wearable Ubiquit. Technol."},{"key":"1_CR26","doi-asserted-by":"crossref","unstructured":"Yang, L., Jiang, D., Xia, X., Pei, E., Oveneke, M.C., Sahli, H.: Multimodal measurement of depression using deep learning models. In: Proceedings of the 7th Annual Workshop on Audio\/Visual Emotion Challenge, pp. 53\u201359 (2017)","DOI":"10.1145\/3133944.3133948"},{"key":"1_CR27","unstructured":"Jain, R.: Improving performance and inference on audio classification tasks using capsule networks. arXiv preprint arXiv:1902.05069 (2019)"},{"key":"1_CR28","doi-asserted-by":"crossref","unstructured":"Chao, L., Tao, J., Yang, M., Li, Y.: Multi task sequence learning for depression scale prediction from video. In: 2015 International Conference on Affective Computing and Intelligent Interaction (ACII), pp. 526\u2013531. IEEE (2015)","DOI":"10.1109\/ACII.2015.7344620"},{"key":"1_CR29","doi-asserted-by":"crossref","unstructured":"Gupta, R., Sahu, S., Espy-Wilson, C.Y., Narayanan, S.S.: An affect prediction approach through depression severity parameter incorporation in neural networks. In: Interspeech, pp. 3122\u20133126 (2017)","DOI":"10.21437\/Interspeech.2017-120"},{"key":"1_CR30","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"13","DOI":"10.1007\/978-3-319-69923-3_2","volume-title":"Biometric Recognition","author":"Y Kang","year":"2017","unstructured":"Kang, Y., Jiang, X., Yin, Y., Shang, Y., Zhou, X.: Deep transformation learning for depression diagnosis from facial images. In: Zhou, J., et al. (eds.) CCBR 2017. LNCS, vol. 10568, pp. 13\u201322. Springer, Cham (2017). https:\/\/doi.org\/10.1007\/978-3-319-69923-3_2"},{"key":"1_CR31","doi-asserted-by":"crossref","unstructured":"Yu, G., Slotine, J.J.: Audio classification from time-frequency texture. In: 2009 IEEE International Conference on Acoustics, Speech and Signal Processing, pp. 1677\u20131680 (2009)","DOI":"10.1109\/ICASSP.2009.4959924"},{"key":"1_CR32","doi-asserted-by":"crossref","unstructured":"Ringeval, F., Sonderegger, A., Sauer, J., Lalanne, D.: Introducing the RECOLA multimodal corpus of remote collaborative and affective interactions. In: 2013 10th IEEE International Conference and Workshops on Automatic Face and Gesture Recognition (FG), pp. 1\u20138. IEEE (2013)","DOI":"10.1109\/FG.2013.6553805"},{"key":"1_CR33","unstructured":"Gratch, J., et al.: The distress analysis interview corpus of human and computer interviews. LREC, pp. 3123\u20133128 (2014)"},{"key":"1_CR34","doi-asserted-by":"crossref","unstructured":"Ma, X., Yang, H., Chen, Q., Huang, D., Wang, Y.: Depaudionet: an efficient deep model for audio based depression classification. In: Proceedings of the 6th International Workshop on Audio\/Visual Emotion Challenge, pp. 35\u201342 (2016)","DOI":"10.1145\/2988257.2988267"},{"key":"1_CR35","unstructured":"Rejaibi, E., Komaty, A., Meriaudeau, F., Agrebi, S., Othmani, A.: MFCC-based recurrent neural network for automatic clinical depression recognition and assessment from speech. arXiv preprint arXiv:1909.07208 (2019)"},{"key":"1_CR36","doi-asserted-by":"crossref","unstructured":"Tzirakis, P., Zhang, J., Schuller, B.W.: End-to-end speech emotion recognition using deep neural networks. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5089\u20135093 (2018)","DOI":"10.1109\/ICASSP.2018.8462677"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition. ICPR International Workshops and Challenges"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-68790-8_1","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,18]],"date-time":"2022-12-18T14:08:51Z","timestamp":1671372531000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-030-68790-8_1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021]]},"ISBN":["9783030687892","9783030687908"],"references-count":36,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-68790-8_1","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021]]},"assertion":[{"value":"23 February 2021","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICPR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Pattern Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2021","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"10 January 2021","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"11 January 2021","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ICPR2020","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.icpr2020.it\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}