{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,9]],"date-time":"2026-05-09T16:50:00Z","timestamp":1778345400558,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":52,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681491","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:33Z","timestamp":1729925973000},"page":"321-329","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":8,"title":["MDDR: Multi-modal Dual-Attention aggregation for Depression Recognition"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-8079-3349","authenticated-orcid":false,"given":"Wei","family":"Zhang","sequence":"first","affiliation":[{"name":"National University of Defense Technology, ChangSha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2305-7555","authenticated-orcid":false,"given":"En","family":"Zhu","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, ChangSha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-8703-7276","authenticated-orcid":false,"given":"Juan","family":"Chen","sequence":"additional","affiliation":[{"name":"University of Chinese Academy of Sciences, BeiJing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-4338-4960","authenticated-orcid":false,"given":"YunPeng","family":"Li","sequence":"additional","affiliation":[{"name":"Nanjing Industria Tenebris Information Technology Co., Ltd, NanJing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0140-6736(21)02141-3"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1016\/S2215-0366(21)00251-0"},{"key":"e_1_3_2_1_3_1","volume-title":"Depression and other common mental disorders: global health estimates","author":"Mathers C","year":"2020","unstructured":"Mathers C, Fat DM, Boerma JT. Depression and other common mental disorders: global health estimates. World Health Organization; 2020."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","unstructured":"Geisser M E Roth R S Robinson M E. Assessing depression among persons with chronic pain using the Center for Epidemiological Studies-Depression Scale and the Beck Depression Inventory: a comparative analysis[J]. The Clinical journal of pain 1997 13(2): 163--170.","DOI":"10.1097\/00002508-199706000-00011"},{"key":"e_1_3_2_1_5_1","article-title":"Automatic assessment of depression based on visual cues: A systematic review","author":"Pampouchidou P.","year":"2017","unstructured":"A. Pampouchidou, P. Simos, K. Marias, F. Meriaudeau, F. Yang, M. Pediaditis, and M. Tsiknakis, 'Automatic assessment of depression based on visual cues: A systematic review,' IEEE Trans. on Affective Computing, 2017.","journal-title":"IEEE Trans. on Affective Computing"},{"key":"e_1_3_2_1_6_1","volume-title":"Non-verbal Communication in Depression","author":"Ellgring","year":"2007","unstructured":"H. Ellgring, Non-verbal Communication in Depression, Cambridge University Press, New York, 2007."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2018.2870884"},{"key":"e_1_3_2_1_8_1","first-page":"4544","article-title":"Depression detection based on deep distribution learning[C]\/\/2019 IEEE international conference on image processing (ICIP)","volume":"2019","author":"De Melo W C","unstructured":"De Melo W C, Granger E, Hadid A. Depression detection based on deep distribution learning[C]\/\/2019 IEEE international conference on image processing (ICIP). IEEE, 2019: 4544--4548.","journal-title":"IEEE"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","unstructured":"Safari P India M Hernando J .Self-attention encoding and pooling for speaker recognition[J]. 2020.DOI:10.48550\/arXiv.2008.01077.","DOI":"10.48550\/arXiv.2008.01077"},{"key":"e_1_3_2_1_10_1","volume-title":"Attention is all you need[J]. Advances in neural information processing systems","author":"Vaswani A","year":"2017","unstructured":"Vaswani A, Shazeer N, Parmar N, et al. Attention is all you need[J]. Advances in neural information processing systems, 2017, 30."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"Hu J Shen L Sun G. Squeeze-and-excitation networks[C]\/\/Proceedings of the IEEE conference on computer vision and pattern recognition. 2018: 7132--7141.","DOI":"10.1109\/CVPR.2018.00745"},{"key":"e_1_3_2_1_12_1","first-page":"6917","article-title":"Multimodal transformer with learnable frontend and self attention for emotion recognition[C]\/\/ICASSP 2022--2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","volume":"2022","author":"Dutta S","unstructured":"Dutta S, Ganapathy S. Multimodal transformer with learnable frontend and self attention for emotion recognition[C]\/\/ICASSP 2022--2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2022: 6917--6921.","journal-title":"IEEE"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"crossref","unstructured":"Lv F Chen X Huang Y et al. Progressive modality reinforcement for human multimodal emotion recognition from unaligned multimodal sequences[ C]\/\/Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2021: 2554--2562.","DOI":"10.1109\/CVPR46437.2021.00258"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"crossref","unstructured":"Zhang Z An L Cui Z et al. Abaw5 challenge: a facial affect recognition approach utilizing transformer encoder and audiovisual fusion[C]\/\/Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2023: 5724- 5733.","DOI":"10.1109\/CVPRW59228.2023.00607"},{"key":"e_1_3_2_1_15_1","volume-title":"CubeMLP: An MLP-based model for multimodal sentiment analysis and depression estimation[C]\/\/Proceedings of the 30th ACM international conference on multimedia. 2022: 3722--3729","author":"Sun H","unstructured":"Sun H, Wang H, Liu J, et al. CubeMLP: An MLP-based model for multimodal sentiment analysis and depression estimation[C]\/\/Proceedings of the 30th ACM international conference on multimedia. 2022: 3722--3729."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2023.102017"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1049\/el.2019.0443"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACII.2015.7344620"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054375"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","unstructured":"Uddin M A Joolee J B Lee Y K .Depression Level Prediction Using Deep Spatiotemporal Features and Multilayer Bi-LTSM[J].IEEE Transactions on Affective Computing 2020 PP(99):1--1.DOI:10.1109\/TAFFC.2020.2970418.","DOI":"10.1109\/TAFFC.2020.2970418"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","unstructured":"Melo W C D Granger E Hadid A .Combining Global and Local Convolutional 3D Networks for Detecting Depression from Facial Expressions[C]\/\/14th IEEE International Conference on Automatic Face and Gesture Recognition.IEEE 2019.DOI:10.1109\/FG.2019.8756568.","DOI":"10.1109\/FG.2019.8756568"},{"key":"e_1_3_2_1_22_1","volume-title":"Measuring depression symptom severity from spoken language and 3D facial expressions[J]. arXiv preprint arXiv:1811.08592","author":"Haque A","year":"2018","unstructured":"Haque A, Guo M, Miner A S, et al. Measuring depression symptom severity from spoken language and 3D facial expressions[J]. arXiv preprint arXiv:1811.08592, 2018."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1002\/int.22426"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","unstructured":"Choi D Zhang G Kim D E et al.Depression Diagnosis Algorithm Based on 2-stream CNN Using Facial Image[C]\/\/2023 IEEE\/ACIS 23rd International Conference on Computer and Information Science (ICIS).0[2024-02- 05].DOI:10.1109\/ICIS57766.2023.10210234.","DOI":"10.1109\/ICIS57766.2023.10210234"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2020.10.015"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3275156"},{"key":"e_1_3_2_1_27_1","volume-title":"Multi-modal depression estimation based on sub-attentional fusion[C]\/\/European Conference on Computer Vision","author":"Wei P C","year":"2022","unstructured":"Wei P C, Peng K, Roitberg A, et al. Multi-modal depression estimation based on sub-attentional fusion[C]\/\/European Conference on Computer Vision. Cham: Springer Nature Switzerland, 2022: 623--639."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2020.2970712"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2017.2650899"},{"key":"e_1_3_2_1_30_1","volume-title":"Avec 2013: the continuous audio\/visual emotion and depression recognition challenge[C]\/\/Proceedings of the 3rd ACM international workshop on Audio\/visual emotion challenge. 2013: 3--10","author":"Valstar M","unstructured":"Valstar M, Schuller B, Smith K, et al. Avec 2013: the continuous audio\/visual emotion and depression recognition challenge[C]\/\/Proceedings of the 3rd ACM international workshop on Audio\/visual emotion challenge. 2013: 3--10."},{"key":"e_1_3_2_1_31_1","volume-title":"Avec 2014: 3d dimensional affect and depression recognition challenge[C]\/\/Proceedings of the 4th international workshop on audio\/visual emotion challenge. 2014: 3--10","author":"Valstar M","unstructured":"Valstar M, Schuller B, Smith K, et al. Avec 2014: 3d dimensional affect and depression recognition challenge[C]\/\/Proceedings of the 4th international workshop on audio\/visual emotion challenge. 2014: 3--10."},{"key":"e_1_3_2_1_32_1","volume-title":"Eyes whisper depression: A CCA based multimodal approach[ C]\/\/Proceedings of the 22nd ACM international conference on Multimedia. 2014: 961--964","author":"Kaya H","unstructured":"Kaya H, Salah A A. Eyes whisper depression: A CCA based multimodal approach[ C]\/\/Proceedings of the 22nd ACM international conference on Multimedia. 2014: 961--964."},{"key":"e_1_3_2_1_33_1","volume-title":"4th InternationalWorkshop on Audio\/Visual Emotion Challenge. 2014: 19--26","author":"Kaya H","unstructured":"Kaya H, \u00c7illi F, Salah A A. Ensemble CCA for continuous emotion prediction[ C]\/\/Proceedings of the 4th InternationalWorkshop on Audio\/Visual Emotion Challenge. 2014: 19--26."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/FG.2018.00019"},{"key":"e_1_3_2_1_35_1","first-page":"6533","article-title":"HuBERT: How much can a bad teacher benefit ASR pre-training?[C]\/\/ICASSP 2021--2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","volume":"2021","author":"Hsu W N","unstructured":"Hsu W N, Tsai Y H H, Bolte B, et al. HuBERT: How much can a bad teacher benefit ASR pre-training?[C]\/\/ICASSP 2021--2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2021: 6533--6537.","journal-title":"IEEE"},{"key":"e_1_3_2_1_36_1","first-page":"6897","article-title":"Key-sparse transformer for multimodal speech emotion recognition[C]\/\/ICASSP 2022--2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","volume":"2022","author":"Chen W","unstructured":"Chen W, Xing X, Xu X, et al. Key-sparse transformer for multimodal speech emotion recognition[C]\/\/ICASSP 2022--2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2022: 6897--6901.","journal-title":"IEEE"},{"key":"e_1_3_2_1_37_1","first-page":"7367","article-title":"Speech emotion recognition with co-attention based multi-level acoustic information[C]\/\/ICASSP 2022--2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","volume":"2022","author":"Zou H","unstructured":"Zou H, Si Y, Chen C, et al. Speech emotion recognition with co-attention based multi-level acoustic information[C]\/\/ICASSP 2022--2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2022: 7367- 7371.","journal-title":"IEEE"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"crossref","unstructured":"Beck AT Steer RA Ball R Ranieri WF. Comparison of beck depression inventories- IA and -II in psychiatric outpatients. J Person Assessment. 1996;67(3):588--597.","DOI":"10.1207\/s15327752jpa6703_13"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1037\/1040-3590.7.1.59"},{"key":"e_1_3_2_1_40_1","first-page":"492","volume-title":"Methodology and distribution","author":"Huber","year":"1992","unstructured":"P. J. Huber, 'Robust estimation of a location parameter,' Breakthroughs in statistics: Methodology and distribution, pp. 492--518, 1992."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2018.2828819"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053762"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2017.2740923"},{"key":"e_1_3_2_1_44_1","first-page":"248","article-title":"Imagenet: A large-scale hierarchical image database[C]\/\/2009 IEEE conference on computer vision and pattern recognition","volume":"2009","author":"Deng J","unstructured":"Deng J, Dong W, Socher R, et al. Imagenet: A large-scale hierarchical image database[C]\/\/2009 IEEE conference on computer vision and pattern recognition. Ieee, 2009: 248--255.","journal-title":"Ieee"},{"key":"e_1_3_2_1_45_1","volume-title":"Depa: Self-supervised audio embedding for depression detection[C]\/\/Proceedings of the 29th ACM international conference on multimedia. 2021: 135--143","author":"Zhang P","unstructured":"Zhang P, Wu M, Dinkel H, et al. Depa: Self-supervised audio embedding for depression detection[C]\/\/Proceedings of the 29th ACM international conference on multimedia. 2021: 135--143."},{"key":"e_1_3_2_1_46_1","volume-title":"Explainable Depression Detection using Multimodal Behavioural Cues[C]\/\/Proceedings of the 25th International Conference on Multimodal Interaction. 2023: 721--725","author":"Gahalawat M.","unstructured":"Gahalawat M. Explainable Depression Detection using Multimodal Behavioural Cues[C]\/\/Proceedings of the 25th International Conference on Multimodal Interaction. 2023: 721--725."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_48_1","volume-title":"Multimodal spatiotemporal representation for automatic depression level detection[J]","author":"Niu M","year":"2020","unstructured":"Niu M, Tao J, Liu B, et al. Multimodal spatiotemporal representation for automatic depression level detection[J]. IEEE transactions on affective computing, 2020, 14(1): 294--307."},{"key":"e_1_3_2_1_49_1","volume-title":"Lopez M B. MDN: A deep maximization-differentiation network for spatio-temporal depression detection[J]","author":"De Melo WC","year":"2021","unstructured":"De MeloWC, Granger E, Lopez M B. MDN: A deep maximization-differentiation network for spatio-temporal depression detection[J]. IEEE transactions on affective computing, 2021, 14(1): 578--590."},{"key":"e_1_3_2_1_50_1","volume-title":"A deep multiscale spatiotemporal network for assessing depression from facial dynamics[J]","author":"De Melo W C","year":"2020","unstructured":"De Melo W C, Granger E, Hadid A. A deep multiscale spatiotemporal network for assessing depression from facial dynamics[J]. IEEE transactions on affective computing, 2020, 13(3): 1581--1592."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2023.3310329"},{"key":"e_1_3_2_1_52_1","volume-title":"Tmac: Temporal multi-modal graph learning for acoustic event classification[C]\/\/Proceedings of the 31st ACM International Conference on Multimedia. 2023: 3365--3374","author":"Liu M","unstructured":"Liu M, Liang K, Hu D, et al. Tmac: Temporal multi-modal graph learning for acoustic event classification[C]\/\/Proceedings of the 31st ACM International Conference on Multimedia. 2023: 3365--3374."}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681491","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681491","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:57:47Z","timestamp":1750294667000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681491"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":52,"alternative-id":["10.1145\/3664647.3681491","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681491","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}