{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T00:25:53Z","timestamp":1784766353438,"version":"3.55.0"},"publisher-location":"Cham","reference-count":48,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031539596","type":"print"},{"value":"9783031539602","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-3-031-53960-2_9","type":"book-chapter","created":{"date-parts":[[2024,3,20]],"date-time":"2024-03-20T05:54:30Z","timestamp":1710914070000},"page":"124-141","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["Audio-Based Detection of\u00a0Anxiety and\u00a0Depression via\u00a0Vocal Biomarkers"],"prefix":"10.1007","author":[{"given":"Raymond","family":"Brueckner","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Namhee","family":"Kwon","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Vinod","family":"Subramanian","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nate","family":"Blaylock","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Henry","family":"O\u2019Connell","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,3,21]]},"reference":[{"key":"9_CR1","unstructured":"Abadi, M., et al.: TensorFlow: large-scale machine learning on heterogeneous systems (2015). Software available from tensorflow.org"},{"key":"9_CR2","doi-asserted-by":"crossref","unstructured":"Arroll, B., et al.: Validation of PHQ-2 and PHQ-9 to screen for major depression in the primary care population. Ann. Family Med. 8(4), 348 (2010)","DOI":"10.1370\/afm.1139"},{"key":"9_CR3","unstructured":"Baevski, A., Zhou, H., Mohamed, A., Auli, M.: Wav2vec 2.0: a framework for self-supervised learning of speech representations. In: Advances in Neural Information Processing Systems, vol. 33 (NeurIPS 2020). Curran Associates Inc., Red Hook, NY, USA (2020)"},{"key":"9_CR4","doi-asserted-by":"publisher","first-page":"327","DOI":"10.31887\/DCNS.2015.17.3\/bbandelow","volume":"17","author":"B Bandelow","year":"2015","unstructured":"Bandelow, B., Michaelis, S.: Epidemiology of anxiety disorders in the 21st century. Dialogues Clin. Neurosci. 17, 327\u2013335 (2015)","journal-title":"Dialogues Clin. Neurosci."},{"issue":"6","key":"9_CR5","doi-asserted-by":"publisher","first-page":"547","DOI":"10.1016\/j.janxdis.2014.06.002","volume":"28","author":"C Beard","year":"2014","unstructured":"Beard, C., Bj\u00f6rgvinsson, T.: Beyond generalized anxiety disorder: psychometric properties of the GAD-7 in a heterogeneous psychiatric sample. J. Anxiety Disord. 28(6), 547\u2013552 (2014)","journal-title":"J. Anxiety Disord."},{"key":"9_CR6","unstructured":"Brueckner, R.: Application of Deep Learning Methods in Computational Paralinguistics. Ph.D. thesis, Technische Universit\u00e4t M\u00fcnchen (2020)"},{"key":"9_CR7","doi-asserted-by":"crossref","unstructured":"Chen, T., Guestrin, C.: XGBoost: a scalable tree boosting system. In: Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining. KDD \u201916, pp. 785\u2013794. ACM, New York (2016)","DOI":"10.1145\/2939672.2939785"},{"issue":"4","key":"9_CR8","doi-asserted-by":"publisher","first-page":"357","DOI":"10.1109\/TASSP.1980.1163420","volume":"28","author":"S Davis","year":"1980","unstructured":"Davis, S., Mermelstein, P.: Comparison of parametric representations for monosyllabic word recognition in continuously spoken sentences. IEEE Trans. Acoust. Speech Sig. Process. 28(4), 357\u2013366 (1980)","journal-title":"IEEE Trans. Acoust. Speech Sig. Process."},{"key":"9_CR9","doi-asserted-by":"crossref","unstructured":"De\u00a0Angel, V., et\u00a0al.: Digital health tools for the passive monitoring of depression: a systematic review of methods. NPJ Digit. Med. 5(1), 3 (2022)","DOI":"10.1038\/s41746-021-00548-8"},{"key":"9_CR10","doi-asserted-by":"crossref","unstructured":"Endler, N.S., Kocovski, N.L.: State and trait anxiety revisited. J. Anxiety Disorders 15(3), 231\u2013245 (2001)","DOI":"10.1016\/S0887-6185(01)00060-3"},{"key":"9_CR11","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-27299-3","volume-title":"Real-Time Speech and Music Classification by Large Audio Feature Space Extraction","author":"F Eyben","year":"2015","unstructured":"Eyben, F.: Real-Time Speech and Music Classification by Large Audio Feature Space Extraction. Springer, Cham (2015). https:\/\/doi.org\/10.1007\/978-3-319-27299-3"},{"key":"9_CR12","doi-asserted-by":"crossref","unstructured":"Eyben, F., et al.: The Geneva minimalistic acoustic parameter set (GeMAPS) for voice research and affective computing. IEEE Trans. Affect. Comput. 7(2), 190\u2013202 (2016)","DOI":"10.1109\/TAFFC.2015.2457417"},{"key":"9_CR13","doi-asserted-by":"crossref","unstructured":"Eyben, F., W\u00f6llmer, M., Schuller, B.: openSMILE \u2013 The Munich versatile and fast open-source audio feature extractor. In: Proceedings of the 18th ACM International Conference on Multimedia, pp. 1459\u20131462. ACM, Florence, Italy (2010)","DOI":"10.1145\/1873951.1874246"},{"issue":"2","key":"9_CR14","doi-asserted-by":"publisher","first-page":"666","DOI":"10.1109\/TAFFC.2019.2944380","volume":"13","author":"Z Huang","year":"2022","unstructured":"Huang, Z., Epps, J., Joachim, D.: Investigation of speech landmark patterns for depression detection. IEEE Trans. Affect. Comput. 13(2), 666\u2013679 (2022)","journal-title":"IEEE Trans. Affect. Comput."},{"key":"9_CR15","unstructured":"Ioffe, S., Szegedy, C.: Batch normalization: accelerating deep network training by reducing internal covariate shift. In: International Conference on Machine Learning, pp. 448\u2013456. PMLR (2015)"},{"key":"9_CR16","doi-asserted-by":"crossref","unstructured":"Jeancolas, L., et al.: X-vectors: new quantitative biomarkers for early Parkinson\u2019s disease detection from speech. Front. Neuroinform. 15, 578369 (2021)","DOI":"10.3389\/fninf.2021.578369"},{"key":"9_CR17","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)"},{"key":"9_CR18","doi-asserted-by":"crossref","unstructured":"Kroenke, K., Spitzer, R.L., Williams, J.B.W.: The PHQ-9: validity of a brief depression severity measure. J. General Internal Med. 16(9), 606\u2013613 (2001)","DOI":"10.1046\/j.1525-1497.2001.016009606.x"},{"key":"9_CR19","doi-asserted-by":"crossref","unstructured":"Ma, X., Yang, H., Chen, Q., Huang, D., Wang, Y.: Depaudionet: an efficient deep model for audio based depression classification. In: Proceedings of the 6th International Workshop on Audio\/visual Emotion Challenge, pp. 35\u201342 (2016)","DOI":"10.1145\/2988257.2988267"},{"key":"9_CR20","doi-asserted-by":"crossref","unstructured":"Moro-Velazquez, L., Villalba, J., Dehak, N.: Using X-vectors to automatically detect Parkinson\u2019s disease from speech. In: ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1155\u20131159. IEEE (2020)","DOI":"10.1109\/ICASSP40776.2020.9053770"},{"key":"9_CR21","unstructured":"Nair, V., Hinton, G.E.: Rectified linear units improve restricted Boltzmann machines. In: International Conference on Machine Learning (ICML) 2010, pp. 807\u2013814 (2010)"},{"key":"9_CR22","doi-asserted-by":"crossref","unstructured":"Nirjhar, E.H., Behzadan, A., Chaspari, T.: Exploring bio-behavioral signal trajectories of state anxiety during public speaking. In: ICASSP 2020 - 2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1294\u20131298. IEEE (2020)","DOI":"10.1109\/ICASSP40776.2020.9054160"},{"key":"9_CR23","doi-asserted-by":"crossref","unstructured":"Pappagari, R., Cho, J., Moro-Velazquez, L., Dehak, N.: Using state of the art speaker recognition and natural language processing technologies to detect Alzheimer\u2019s disease and assess its severity. In: INTERSPEECH, pp. 2177\u20132181 (2020)","DOI":"10.21437\/Interspeech.2020-2587"},{"key":"9_CR24","doi-asserted-by":"crossref","unstructured":"Pappagari, R., Wang, T., Villalba, J., Chen, N., Dehak, N.: X-vectors meet emotions: a study on dependencies between emotion and speaker recognition. In: ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7169\u20137173. IEEE (2020)","DOI":"10.1109\/ICASSP40776.2020.9054317"},{"key":"9_CR25","doi-asserted-by":"publisher","DOI":"10.1016\/j.dsp.2021.103286","volume":"120","author":"Luis Felipe Parra-Gallego and Juan Rafael Orozco-Arroyave","year":"2022","unstructured":"Luis Felipe Parra-Gallego and Juan Rafael Orozco-Arroyave: Classification of emotions and evaluation of customer satisfaction from speech in real world acoustic environments. Digit. Sig. Process. 120, 103286 (2022)","journal-title":"Digit. Sig. Process."},{"key":"9_CR26","first-page":"2825","volume":"12","author":"F Pedregosa","year":"2011","unstructured":"Pedregosa, F., et al.: Scikit-learn: machine Learning in Python. J. Mach. Learn. Res. 12, 2825\u20132830 (2011)","journal-title":"J. Mach. Learn. Res."},{"key":"9_CR27","unstructured":"Povey, D., et\u00a0al.: The Kaldi speech recognition toolkit. In: IEEE 2011 Workshop on Automatic Speech Recognition and Understanding, number CONF. IEEE Signal Processing Society (2011)"},{"key":"9_CR28","doi-asserted-by":"crossref","unstructured":"Raj, D., Snyder, D., Povey, D., Khudanpur, S.: Probing the information encoded in x-vectors. In: 2019 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU), pp. 726\u2013733. IEEE (2019)","DOI":"10.1109\/ASRU46091.2019.9003979"},{"key":"9_CR29","doi-asserted-by":"crossref","unstructured":"Ringeval, F., et al.: AV+EC 2015: the first affect recognition challenge bridging across audio, video, and physiological data. In: Proceedings of the 5th International Workshop on Audio\/Visual Emotion Challenge, pp. 3\u20138. ACM, Brisbane, Australia (2015)","DOI":"10.1145\/2808196.2811642"},{"key":"9_CR30","unstructured":"Sakib, Md.N., Nirjhar, E.H., Feng, K., Behzadan, A., Chaspari, T., Chaspari, T.: Exploring individual differences of public speaking anxiety in real-life and virtual presentations. IEEE Trans. Affect. Comput. 1 (2021)"},{"key":"9_CR31","doi-asserted-by":"crossref","unstructured":"Salekin, A., Eberle, J.W., Glenn, J.J., Teachman, B.A., Stankovic, J.A.: A weakly supervised learning framework for detecting social anxiety and depression. Proc. ACM Interact. Mob. Wearable Ubiquit. Technol. 2(2), 1\u201326 (2018)","DOI":"10.1145\/3214284"},{"key":"9_CR32","unstructured":"Schuller, B.: Intelligent Audio Analysis \u2013 Speech, Music, and Sound Recognition in Real-Life Conditions. Habilitation thesis, Technische Universit\u00e4t M\u00fcnchen, Munich, Germany (2012)"},{"key":"9_CR33","doi-asserted-by":"crossref","unstructured":"Schuller, B., Batliner, A.: Computational Paralinguistics: Emotion, Affect and Personality in Speech and Language Processing. Wiley, Chichester (2014)","DOI":"10.1002\/9781118706664"},{"key":"9_CR34","doi-asserted-by":"crossref","unstructured":"Schuller, B., Steidl, S., Batliner, A.: The INTERSPEECH 2009 emotion challenge. In: Proceedings of the 10th Annual Conference of the International Speech Communication Association (INTERSPEECH). ISCA, Brighton, UK (2009)","DOI":"10.21437\/Interspeech.2009-103"},{"key":"9_CR35","doi-asserted-by":"crossref","unstructured":"Schuller, B., et al.: The INTERSPEECH 2014 computational paralinguistics challenge: cognitive & physical load. In: Proceedings of the 15th Annual Conference of the International Speech Communication Association (INTERSPEECH), Singapore (2014)","DOI":"10.21437\/Interspeech.2014-104"},{"key":"9_CR36","doi-asserted-by":"crossref","unstructured":"Schuller, B., et al.: Affective and behavioural computing: lessons learnt from the first computational paralinguistics challenge. Comput. Speech Lang. 53, 156\u2013180 (2019)","DOI":"10.1016\/j.csl.2018.02.004"},{"key":"9_CR37","doi-asserted-by":"crossref","unstructured":"Schuller, B.W., et al.: The INTERSPEECH 2016 computational paralinguistics challenge: deception, sincerity & native language. In: Proceedings of the 17th Annual Conference of the International Speech Communication Association (INTERSPEECH), vol. 2016, pp. 2001\u20132005. ISCA, San Francisco, CA, USA (2016)","DOI":"10.21437\/Interspeech.2016-129"},{"key":"9_CR38","doi-asserted-by":"crossref","unstructured":"Schuller, B.W., et al.: The INTERSPEECH 2021 computational paralinguistics challenge: COVID-19 cough, COVID-19 speech, escalation & primates. In: Proceedings of the 22nd Annual Conference of the International Speech Communication Association (INTERSPEECH), pp. 431\u2013435 (2021)","DOI":"10.21437\/Interspeech.2021-19"},{"key":"9_CR39","doi-asserted-by":"crossref","unstructured":"Snyder, D., Garcia-Romero, D., McCree, A., Sell, G., Povey, D., Khudanpur, S.: Spoken language recognition using x-vectors. In: Odyssey, pp. 105\u2013111 (2018)","DOI":"10.21437\/Odyssey.2018-15"},{"key":"9_CR40","doi-asserted-by":"crossref","unstructured":"Snyder, D., Garcia-Romero, D., Sell, G., Povey, D., Khudanpur, S.: X-vectors: robust DNN embeddings for speaker recognition. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5329\u20135333. IEEE (2018)","DOI":"10.1109\/ICASSP.2018.8461375"},{"key":"9_CR41","doi-asserted-by":"crossref","unstructured":"Spitzer, R.L., Kroenke, K., Williams, J.B.W., L\u00f6we, B.: A brief measure for assessing generalized anxiety disorder: the GAD-7. Arch. Intern. Med. 166(10), 1092\u20131097 (2006)","DOI":"10.1001\/archinte.166.10.1092"},{"key":"9_CR42","doi-asserted-by":"publisher","unstructured":"Ting, K.M.: Precision and recall. In: Sammut, C., Webb, G.I. (eds.) Encyclopedia of Machine Learning, p. 781. Springer, Boston (2010). https:\/\/doi.org\/10.1007\/978-0-387-30164-8_652","DOI":"10.1007\/978-0-387-30164-8_652"},{"key":"9_CR43","unstructured":"Valstar, M.F., Gratch, J., Schuller, B.W., Ringeval, F., Cowie, R., Pantic, M. (eds.) Proceedings of the 6th International Workshop on Audio\/Visual Emotion Challenge, AVEC@MM 2016. ACM, Amsterdam, October 2016"},{"key":"9_CR44","doi-asserted-by":"crossref","unstructured":"Valstar, M.F., et al.: AVEC 2013: the continuous audio\/visual emotion and depression recognition challenge. In: Schuller, B.W., Valstar, M.F., Cowie, R., Krajewski, J., Pantic, M. (eds.) Proceedings of the 3rd ACM International Workshop on Audio\/Visual Emotion Challenge, AVEC@ACM Multimedia 2013, Barcelona, Spain, 21 October 2013, pp. 3\u201310. ACM (2013)","DOI":"10.1145\/2512530.2512533"},{"key":"9_CR45","doi-asserted-by":"crossref","unstructured":"Waibel, A.H., Hanazawa, T., Hinton, G.E., Shikano, K., Kevin, J.L.: Phoneme recognition using time-delay neural networks. IEEE Trans. Acoustics Speech Sig. Process. 37, 328\u2013339 (1989)","DOI":"10.1109\/29.21701"},{"key":"9_CR46","doi-asserted-by":"crossref","unstructured":"Weninger, F., Eyben, F., Schuller, B.W., Mortillaro, M., Scherer, K.R.: On the acoustics of emotion in audio: what speech, music, and sound have in common. Front. Psychol. 4 (2013)","DOI":"10.3389\/fpsyg.2013.00292"},{"key":"9_CR47","doi-asserted-by":"crossref","unstructured":"Werneck, A.O., Silva, D.R.: Population density, depressive symptoms, and suicidal thoughts. Revista Brasileira de Psiquiatria (2020)","DOI":"10.1590\/1516-4446-2019-0541"},{"key":"9_CR48","unstructured":"Yin, W., Levis, B., Riehm, K.E., et\u00a0al.: Equivalency of the diagnostic accuracy of the PHQ-8 and PHQ-9: a systematic review and individual participant data meta-analysis. Psychol. Med. 50(8), 1368\u20131380 (2020)"}],"container-title":["Lecture Notes in Networks and Systems","Advances in Information and Communication"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-53960-2_9","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,20]],"date-time":"2024-03-20T05:57:05Z","timestamp":1710914225000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-53960-2_9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9783031539596","9783031539602"],"references-count":48,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-53960-2_9","relation":{},"ISSN":["2367-3370","2367-3389"],"issn-type":[{"value":"2367-3370","type":"print"},{"value":"2367-3389","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"21 March 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"FICC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Future of Information and Communication Conference","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Berlin","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Germany","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 April 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"5 April 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ficc2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/saiconference.com\/FICC","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}