{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T04:57:40Z","timestamp":1781585860698,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":56,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"name":"the MUR PNRR project FAIR","award":["PE00000013"],"award-info":[{"award-number":["PE00000013"]}]},{"name":"the NextGenerationEU and by the PRIN project CREATIVE","award":["Prot. 2020ZSL9F9"],"award-info":[{"award-number":["Prot. 2020ZSL9F9"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681443","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:41Z","timestamp":1729925981000},"page":"8189-8198","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":11,"title":["Towards End-to-End Explainable Facial Action Unit Recognition via Vision-Language Joint Learning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3925-4951","authenticated-orcid":false,"given":"Xuri","family":"Ge","sequence":"first","affiliation":[{"name":"University of Glasgow, Glasgow, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4759-2042","authenticated-orcid":false,"given":"Junchen","family":"Fu","sequence":"additional","affiliation":[{"name":"University of Glasgow, Glasgow, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5441-5998","authenticated-orcid":false,"given":"Fuhai","family":"Chen","sequence":"additional","affiliation":[{"name":"Fuzhou University, Fuzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7796-6952","authenticated-orcid":false,"given":"Shan","family":"An","sequence":"additional","affiliation":[{"name":"Tianjin University, Tianjin, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6597-7248","authenticated-orcid":false,"given":"Nicu","family":"Sebe","sequence":"additional","affiliation":[{"name":"University of Trento, Trento, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9228-1759","authenticated-orcid":false,"given":"Joemon M.","family":"Jose","sequence":"additional","affiliation":[{"name":"University of Glasgow, Glasgow, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"crossref","unstructured":"Yanan Chang and Shangfei Wang. 2022. Knowledge-driven self-supervised representation learning for facial action unit recognition. In CVPR. 20417--20426.","DOI":"10.1109\/CVPR52688.2022.01977"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i1.19914"},{"key":"e_1_3_2_1_3_1","first-page":"14338","article-title":"Knowledge augmented deep neural networks for joint facial expression and action unit recognition","volume":"33","author":"Cui Zijun","year":"2020","unstructured":"Zijun Cui, Tengfei Song, Yuru Wang, and Qiang Ji. 2020. Knowledge augmented deep neural networks for joint facial expression and action unit recognition. NeurIPS, Vol. 33 (2020), 14338--14349.","journal-title":"NeurIPS"},{"key":"e_1_3_2_1_4_1","volume-title":"Improved regularization of convolutional neural networks with cutout. arXiv preprint arXiv:1708.04552","author":"DeVries Terrance","year":"2017","unstructured":"Terrance DeVries and Graham W Taylor. 2017. Improved regularization of convolutional neural networks with cutout. arXiv preprint arXiv:1708.04552 (2017)."},{"key":"e_1_3_2_1_5_1","volume-title":"What the face reveals: Basic and applied studies of spontaneous expression using the Facial Action Coding System (FACS)","author":"Ekman Paul","unstructured":"Paul Ekman and Erika L Rosenberg. 1997. What the face reveals: Basic and applied studies of spontaneous expression using the Facial Action Coding System (FACS). Oxford University Press, USA."},{"key":"e_1_3_2_1_6_1","volume-title":"Detection of deception in adults and children via facial expressions. Child Development","author":"Feldman Robert S","year":"1979","unstructured":"Robert S Feldman, Larry Jenkins, and Oladeji Popoola. 1979. Detection of deception in adults and children via facial expressions. Child Development (1979), 350--355."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657725"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3616855.3635805"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475634"},{"key":"e_1_3_2_1_10_1","volume-title":"ALGRNet: Multi-relational adaptive facial action unit modelling for face representation and relevant recognitions","author":"Ge Xuri","year":"2023","unstructured":"Xuri Ge, Joemon M Jose, Pengcheng Wang, Arunachalam Iyer, Xiao Liu, and Hu Han. 2023. ALGRNet: Multi-relational adaptive facial action unit modelling for face representation and relevant recognitions. IEEE Transactions on Biometrics, Behavior, and Identity Science (2023)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3643863"},{"key":"e_1_3_2_1_12_1","volume-title":"Local global relational network for facial action units recognition","author":"Ge Xuri","unstructured":"Xuri Ge, Pengcheng Wan, Hu Han, Joemon M Jose, Zhilong Ji, Zhongqin Wu, and Xiao Liu. 2021. Local global relational network for facial action units recognition. In FG. IEEE, 01--08."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2005.06.042"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"crossref","unstructured":"Yan Huang Qi Wu Chunfeng Song and Liang Wang. 2018. Learning semantic concepts and order for image and sentence matching. In CVPR. 6163--6171.","DOI":"10.1109\/CVPR.2018.00645"},{"key":"e_1_3_2_1_16_1","volume-title":"Facial Action Unit Detection With Transformers","author":"Jacob Geethu Miriam","unstructured":"Geethu Miriam Jacob and Bjorn Stenger. 2021. Facial Action Unit Detection With Transformers. In IEEE CVPR. 7680--7689."},{"key":"e_1_3_2_1_17_1","unstructured":"Xincheng Ju Dong Zhang Rong Xiao Junhui Li Shoushan Li Min Zhang and Guodong Zhou. 2021. Joint multi-modal aspect-sentiment analysis with auxiliary cross-modal relation detection. In EMNLP. 4395--4405."},{"key":"e_1_3_2_1_18_1","unstructured":"Dan Klein and Christopher D Manning. 2003. A* parsing: Fast exact Viterbi parse selection. In NAACL. 119--126."},{"key":"e_1_3_2_1_19_1","volume-title":"Deep learning. nature","author":"LeCun Yann","year":"2015","unstructured":"Yann LeCun, Yoshua Bengio, and Geoffrey Hinton. 2015. Deep learning. nature, Vol. 521, 7553 (2015), 436--444."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"crossref","unstructured":"Guanbin Li Xin Zhu Yirui Zeng Qing Wang and Liang Lin. 2019. Semantic relationships guided representation learning for facial action unit recognition. In AAAI. 8594--8601.","DOI":"10.1609\/aaai.v33i01.33018594"},{"key":"e_1_3_2_1_21_1","volume-title":"Action unit detection with region adaptation, multi-labeling learning and optimal temporal fusing","author":"Li Wei","year":"1841","unstructured":"Wei Li, Farnaz Abtahi, and Zhigang Zhu. 2017. Action unit detection with region adaptation, multi-labeling learning and optimal temporal fusing. In IEEE CVPR. 1841--1850."},{"key":"e_1_3_2_1_22_1","unstructured":"Xiaotian Li Xiang Zhang Taoyue Wang and Lijun Yin. 2023. Knowledge-Spreader: Learning Semi-Supervised Facial Action Dynamics by Consistifying Knowledge Granularity. In ICCV. 20979--20989."},{"key":"e_1_3_2_1_23_1","volume-title":"Disagreement Matters: Exploring Internal Diversification for Redundant Attention in Generic Facial Action Analysis","author":"Li Xiaotian","year":"2023","unstructured":"Xiaotian Li, Zheng Zhang, Xiang Zhang, Taoyue Wang, Zhihua Li, Huiyuan Yang, Umur Ciftci, Qiang Ji, Jeffrey Cohn, and Lijun Yin. 2023. Disagreement Matters: Exploring Internal Diversification for Redundant Attention in Generic Facial Action Analysis. IEEE Trans. Affect. Comput. (2023)."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2013.2253477"},{"key":"e_1_3_2_1_25_1","volume-title":"Network in network. arXiv preprint arXiv:1312.4400","author":"Lin Min","year":"2013","unstructured":"Min Lin, Qiang Chen, and Shuicheng Yan. 2013. Network in network. arXiv preprint arXiv:1312.4400 (2013)."},{"key":"e_1_3_2_1_26_1","volume-title":"Ivor Wai-Hung Tsang, Zibo Meng, Shizhong Han, and Yan Tong.","author":"Liu Ping","year":"2014","unstructured":"Ping Liu, Joey Tianyi Zhou, Ivor Wai-Hung Tsang, Zibo Meng, Shizhong Han, and Yan Tong. 2014. Feature disentangling machine-a novel approach of feature selection and disentangling in facial expression analysis. In ECCV. 151--166."},{"key":"e_1_3_2_1_27_1","volume-title":"Relation modeling with graph convolutional networks for facial action unit detection","author":"Liu Zhilei","unstructured":"Zhilei Liu, Jiahui Dong, Cuicui Zhang, Longbiao Wang, and Jianwu Dang. 2020. Relation modeling with graph convolutional networks for facial action unit detection. In MMM. Springer, 489--501."},{"key":"e_1_3_2_1_28_1","volume-title":"Swin transformer: Hierarchical vision transformer using shifted windows","author":"Liu Ze","unstructured":"Ze Liu, Yutong Lin, Yue Cao, Han Hu, Yixuan Wei, Zheng Zhang, Stephen Lin, and Baining Guo. 2021. Swin transformer: Hierarchical vision transformer using shifted windows. In IEEE ICCV. 10012--10022."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10489-021-02464-6"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"crossref","unstructured":"Cheng Luo Siyang Song Weicheng Xie Linlin Shen and Hatice Gunes. 2022. Learning Multi-dimensional Edge Feature-based AU Relation Graph for Facial Action Unit Recognition. In IJCAI. 1239--1246.","DOI":"10.24963\/ijcai.2022\/173"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/T-AFFC.2013.4"},{"key":"e_1_3_2_1_32_1","volume-title":"Knowledge Graph Cross-View Contrastive Learning for Recommendation. In European Conference on Information Retrieval. Springer, 3--18","author":"Meng Zeyuan","year":"2024","unstructured":"Zeyuan Meng, Iadh Ounis, Craig Macdonald, and Zixuan Yi. 2024. Knowledge Graph Cross-View Contrastive Learning for Recommendation. In European Conference on Information Retrieval. Springer, 3--18."},{"key":"e_1_3_2_1_33_1","volume-title":"Local relationship learning with person-specific shape regularization for facial action unit detection","author":"Niu Xuesong","year":"1917","unstructured":"Xuesong Niu, Hu Han, Songfan Yang, Yan Huang, and Shiguang Shan. 2019. Local relationship learning with person-specific shape regularization for facial action unit detection. In IEEE CVPR. 11917--11926."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10489-022-03463-x"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2019.10.076"},{"key":"e_1_3_2_1_36_1","volume-title":"Swinface: a multi-task transformer for face recognition, expression recognition, age estimation and attribute estimation","author":"Qin Lixiong","year":"2023","unstructured":"Lixiong Qin, Mei Wang, Chao Deng, Ke Wang, Xi Chen, Jiani Hu, and Weihong Deng. 2023. Swinface: a multi-task transformer for face recognition, expression recognition, age estimation and attribute estimation. IEEE Transactions on Circuits and Systems for Video Technology (2023)."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1016\/0006-3223(92)90120-O"},{"key":"e_1_3_2_1_38_1","volume-title":"A survey on oversmoothing in graph neural networks. arXiv preprint arXiv:2303.10993","author":"Rusch T Konstantin","year":"2023","unstructured":"T Konstantin Rusch, Michael M Bronstein, and Siddhartha Mishra. 2023. A survey on oversmoothing in graph neural networks. arXiv preprint arXiv:2303.10993 (2023)."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"crossref","unstructured":"Zhiwen Shao Zhilei Liu Jianfei Cai and Lizhuang Ma. 2018. Deep adaptive attention for joint facial action unit detection and face alignment. In ECCV. 705--720.","DOI":"10.1007\/978-3-030-01261-8_43"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-020-01378-z"},{"key":"e_1_3_2_1_41_1","volume-title":"Facial action unit detection using attention and relation learning","author":"Shao Zhiwen","year":"2019","unstructured":"Zhiwen Shao, Zhilei Liu, Jianfei Cai, Yunsheng Wu, and Lizhuang Ma. 2019. Facial action unit detection using attention and relation learning. IEEE Trans. Affect. Comput. (2019)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2023.3277794"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"crossref","unstructured":"Tengfei Song Lisha Chen Wenming Zheng and Qiang Ji. 2021. Uncertain graph neural networks for facial action unit detection. In AAAI. 5993--6001.","DOI":"10.1609\/aaai.v35i7.16748"},{"key":"e_1_3_2_1_44_1","volume-title":"Hybrid Message Passing With Performance-Driven Structures for Facial Action Unit Detection","author":"Song Tengfei","unstructured":"Tengfei Song, Zijun Cui, Wenming Zheng, and Qiang Ji. 2021. Hybrid Message Passing With Performance-Driven Structures for Facial Action Unit Detection. In IEEE CVPR. 6267--6276."},{"key":"e_1_3_2_1_45_1","volume-title":"Phocnet: A deep convolutional neural network for word spotting in handwritten documents","author":"Sudholt Sebastian","year":"2016","unstructured":"Sebastian Sudholt and Gernot A Fink. 2016. Phocnet: A deep convolutional neural network for word spotting in handwritten documents. In ICFHR. IEEE, 277--282."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2014.2331141"},{"key":"e_1_3_2_1_47_1","volume-title":"Learning bayesian networks with qualitative constraints","author":"Tong Yan","unstructured":"Yan Tong and Qiang Ji. 2008. Learning bayesian networks with qualitative constraints. In IEEE CVPR. 1--8."},{"key":"e_1_3_2_1_48_1","article-title":"Visualizing data using t-SNE","volume":"9","author":"der Maaten Laurens Van","year":"2008","unstructured":"Laurens Van der Maaten and Geoffrey Hinton. 2008. Visualizing data using t-SNE. Journal of machine learning research, Vol. 9, 11 (2008).","journal-title":"Journal of machine learning research"},{"key":"e_1_3_2_1_49_1","volume-title":"Cbam: Convolutional block attention module. In ECCV. 3--19.","author":"Woo Sanghyun","year":"2018","unstructured":"Sanghyun Woo, Jongchan Park, Joon-Young Lee, and In So Kweon. 2018. Cbam: Convolutional block attention module. In ECCV. 3--19."},{"key":"e_1_3_2_1_50_1","volume-title":"ICML. PMLR","author":"Xu Kelvin","year":"2015","unstructured":"Kelvin Xu, Jimmy Ba, Ryan Kiros, Kyunghyun Cho, Aaron Courville, Ruslan Salakhudinov, Rich Zemel, and Yoshua Bengio. 2015. Show, attend and tell: Neural image caption generation with visual attention. In ICML. PMLR, 2048--2057."},{"key":"e_1_3_2_1_51_1","volume-title":"Exploiting Semantic Embedding and Visual Feature for Facial Action Unit Detection","author":"Yang Huiyuan","unstructured":"Huiyuan Yang, Lijun Yin, Yi Zhou, and Jiuxiang Gu. 2021. Exploiting Semantic Embedding and Visual Feature for Facial Action Unit Detection. In IEEE CVPR. 10482--10491."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/3578932"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.imavis.2014.06.002"},{"key":"e_1_3_2_1_54_1","volume-title":"Jeffrey F Cohn, and Honggang Zhang.","author":"Zhao Kaili","year":"2015","unstructured":"Kaili Zhao, Wen-Sheng Chu, Fernando De la Torre, Jeffrey F Cohn, and Honggang Zhang. 2015. Joint patch and multi-label learning for facial action unit detection. In IEEE CVPR. 2207--2216."},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2016.2570550"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"crossref","unstructured":"Fengda Zhu Yi Zhu Xiaojun Chang and Xiaodan Liang. 2020. Vision-language navigation with self-supervised auxiliary reasoning tasks. In CVPR. 10012--10022.","DOI":"10.1109\/CVPR42600.2020.01003"}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681443","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681443","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:57:47Z","timestamp":1750294667000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681443"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":56,"alternative-id":["10.1145\/3664647.3681443","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681443","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}