{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T15:49:51Z","timestamp":1781797791291,"version":"3.54.5"},"publisher-location":"Cham","reference-count":77,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031729720","type":"print"},{"value":"9783031729737","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72973-7_25","type":"book-chapter","created":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T14:03:04Z","timestamp":1730383384000},"page":"427-445","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":23,"title":["AUFormer: Vision Transformers Are Parameter-Efficient Facial Action Unit Detectors"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-2353-2436","authenticated-orcid":false,"given":"Kaishen","family":"Yuan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0422-6616","authenticated-orcid":false,"given":"Zitong","family":"Yu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2242-6139","authenticated-orcid":false,"given":"Xin","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8946-7472","authenticated-orcid":false,"given":"Weicheng","family":"Xie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2517-9783","authenticated-orcid":false,"given":"Huanjing","family":"Yue","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7521-7920","authenticated-orcid":false,"given":"Jingyu","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,11,1]]},"reference":[{"key":"25_CR1","unstructured":"Ba, J.L., Kiros, J.R., Hinton, G.E.: Layer normalization. arXiv preprint arXiv:1607.06450 (2016)"},{"key":"25_CR2","unstructured":"Bresson, X., Laurent, T.: Residual gated graph convnets. arXiv preprint arXiv:1711.07553 (2017)"},{"key":"25_CR3","doi-asserted-by":"crossref","unstructured":"Chang, Y., Wang, S.: Knowledge-driven self-supervised representation learning for facial action unit recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 20417\u201320426 (2022)","DOI":"10.1109\/CVPR52688.2022.01977"},{"key":"25_CR4","unstructured":"Chen, S., et al.: Adaptformer: Adapting vision transformers for scalable visual recognition. Adv. Neural. Inf. Process. Syst. 35, 16664\u201316678 (2022)"},{"key":"25_CR5","doi-asserted-by":"crossref","unstructured":"Cui, Z., Kuang, C., Gao, T., Talamadupula, K., Ji, Q.: Biomechanics-guided facial action unit detection through force modeling. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8694\u20138703 (2023)","DOI":"10.1109\/CVPR52729.2023.00840"},{"key":"25_CR6","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: Imagenet: a large-scale hierarchical image database. In: 2009 IEEE conference on computer vision and pattern recognition, pp. 248\u2013255. IEEE (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"25_CR7","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"25_CR8","unstructured":"Dosovitskiy, A., et\u00a0al.: An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"25_CR9","doi-asserted-by":"crossref","unstructured":"Ekman, P., Friesen, W.V.: Facial action coding system. Environmental Psychology and Nonverbal Behavior (1978)","DOI":"10.1037\/t27734-000"},{"key":"25_CR10","doi-asserted-by":"crossref","unstructured":"Ginosar, S., Rakelly, K., Sachs, S., Yin, B., Efros, A.A.: A century of portraits: a visual historical record of american high school yearbooks. In: Proceedings of the IEEE International Conference on Computer Vision Workshops, pp.\u00a01\u20137 (2015)","DOI":"10.1109\/ICCVW.2015.87"},{"key":"25_CR11","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"25_CR12","unstructured":"Hendrycks, D., Gimpel, K.: Gaussian error linear units (gelus). arXiv preprint arXiv:1606.08415 (2016)"},{"key":"25_CR13","unstructured":"Houlsby, N., et al.: Parameter-efficient transfer learning for NLP. In: International Conference on Machine Learning, pp. 2790\u20132799. PMLR (2019)"},{"key":"25_CR14","unstructured":"Hu, E.J., et al.: Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685 (2021)"},{"key":"25_CR15","unstructured":"Jacob, G.M., Stenger, B.: Facial action unit detection with transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7680\u20137689 (2021)"},{"key":"25_CR16","doi-asserted-by":"publisher","unstructured":"Jia, M., et al.: Visual prompt tuning. In: European Conference on Computer Vision, pp. 709\u2013727. Springer (2022). https:\/\/doi.org\/10.1007\/978-3-031-19827-4_41","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"25_CR17","doi-asserted-by":"crossref","unstructured":"Jiang, B., Valstar, M.F., Pantic, M.: Action unit detection using sparse appearance descriptors in space-time video volumes. In: 2011 IEEE International Conference on Automatic Face and Gesture Recognition (FG), pp. 314\u2013321. IEEE (2011)","DOI":"10.1109\/FG.2011.5771416"},{"key":"25_CR18","unstructured":"Jie, S., Deng, Z.H.: Convolutional bypasses are better vision transformer adapters. arXiv preprint arXiv:2207.07039 (2022)"},{"key":"25_CR19","doi-asserted-by":"crossref","unstructured":"Karras, T., Laine, S., Aila, T.: A style-based generator architecture for generative adversarial networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4401\u20134410 (2019)","DOI":"10.1109\/CVPR.2019.00453"},{"key":"25_CR20","doi-asserted-by":"crossref","unstructured":"Karras, T., Laine, S., Aittala, M., Hellsten, J., Lehtinen, J., Aila, T.: Analyzing and improving the image quality of stylegan. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8110\u20138119 (2020)","DOI":"10.1109\/CVPR42600.2020.00813"},{"key":"25_CR21","doi-asserted-by":"crossref","unstructured":"Klaser, A., Marsza\u0142ek, M., Schmid, C.: A spatio-temporal descriptor based on 3D-gradients. In: BMVC 2008-19th British Machine Vision Conference, pp. 275\u20131. British Machine Vision Association (2008)","DOI":"10.5244\/C.22.99"},{"key":"25_CR22","doi-asserted-by":"publisher","unstructured":"Kuang, C., Cui, Z., Kephart, J.O., Ji, Q.: Au-aware 3D face reconstruction through personalized au-specific blendshape learning. In: European Conference on Computer Vision, pp. 1\u201318. Springer (2022). https:\/\/doi.org\/10.1007\/978-3-031-19778-9_1","DOI":"10.1007\/978-3-031-19778-9_1"},{"key":"25_CR23","doi-asserted-by":"crossref","unstructured":"Li, G., Zhu, X., Zeng, Y., Wang, Q., Lin, L.: Semantic relationships guided representation learning for facial action unit recognition. In: Proceedings of the AAAI Conference on Artificial Intelligence. vol.\u00a033, pp. 8594\u20138601 (2019)","DOI":"10.1609\/aaai.v33i01.33018594"},{"issue":"11","key":"25_CR24","doi-asserted-by":"publisher","first-page":"2583","DOI":"10.1109\/TPAMI.2018.2791608","volume":"40","author":"W Li","year":"2018","unstructured":"Li, W., Abtahi, F., Zhu, Z., Yin, L.: Eac-net: deep nets with enhancing and cropping for facial action unit detection. IEEE Trans. Pattern Anal. Mach. Intell. 40(11), 2583\u20132596 (2018)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"25_CR25","doi-asserted-by":"crossref","unstructured":"Li, X., Zhang, X., Wang, T., Yin, L.: Knowledge-spreader: Learning semi-supervised facial action dynamics by consistifying knowledge granularity. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 20979\u201320989 (2023)","DOI":"10.1109\/ICCV51070.2023.01918"},{"key":"25_CR26","doi-asserted-by":"publisher","first-page":"221","DOI":"10.1016\/j.neucom.2021.01.032","volume":"436","author":"Y Li","year":"2021","unstructured":"Li, Y., Huang, X., Zhao, G.: Micro-expression action unit detection with spatial and channel attention. Neurocomputing 436, 221\u2013231 (2021)","journal-title":"Neurocomputing"},{"key":"25_CR27","doi-asserted-by":"crossref","unstructured":"Li, Y., Peng, W., Zhao, G.: Micro-expression action unit detection with dual-view attentive similarity-preserving knowledge distillation. In: 2021 16th IEEE International Conference on Automatic Face and Gesture Recognition (FG 2021), pp. 01\u201308. IEEE (2021)","DOI":"10.1109\/FG52635.2021.9666975"},{"key":"25_CR28","doi-asserted-by":"crossref","unstructured":"Li, Y., Zhao, G.: Intra-and inter-contrastive learning for micro-expression action unit detection. In: Proceedings of the 2021 International Conference on Multimodal Interaction, pp. 702\u2013706 (2021)","DOI":"10.1145\/3462244.3479956"},{"key":"25_CR29","unstructured":"Li, Y., Tarlow, D., Brockschmidt, M., Zemel, R.: Gated graph sequence neural networks. arXiv preprint arXiv:1511.05493 (2015)"},{"key":"25_CR30","doi-asserted-by":"crossref","unstructured":"Liu, X., et al.: Multi-scale promoted self-adjusting correlation learning for facial action unit detection. arXiv preprint arXiv:2308.07770 (2023)","DOI":"10.1109\/TAFFC.2024.3460538"},{"key":"25_CR31","doi-asserted-by":"crossref","unstructured":"Liu, X., Zhang, Y., Yu, Z., Lu, H., Yue, H., Yang, J.: RPPG-mae: self-supervised pretraining with masked autoencoders for remote physiological measurements. IEEE Trans. Multi. (2024)","DOI":"10.1109\/TMM.2024.3363660"},{"key":"25_CR32","doi-asserted-by":"crossref","unstructured":"Liu, Z., et al.: Swin transformer: hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10012\u201310022 (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"25_CR33","doi-asserted-by":"crossref","unstructured":"Liu, Z., Luo, P., Wang, X., Tang, X.: Deep learning face attributes in the wild. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 3730\u20133738 (2015)","DOI":"10.1109\/ICCV.2015.425"},{"key":"25_CR34","doi-asserted-by":"crossref","unstructured":"Lu, H., et\u00a0al.: GPT as psychologist? Preliminary evaluations for gpt-4v on visual affective computing. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 322\u2013331 (2024)","DOI":"10.1109\/CVPRW63382.2024.00037"},{"key":"25_CR35","doi-asserted-by":"crossref","unstructured":"Luo, C., Song, S., Xie, W., Shen, L., Gunes, H.: Learning multi-dimensional edge feature-based au relation graph for facial action unit recognition. arXiv preprint arXiv:2205.01782 (2022)","DOI":"10.24963\/ijcai.2022\/173"},{"key":"25_CR36","unstructured":"Ma, B., et al.: Facial action unit detection and intensity estimation from self-supervised representation. arXiv preprint arXiv:2210.15878 (2022)"},{"key":"25_CR37","unstructured":"Van\u00a0der Maaten, L., Hinton, G.: Visualizing data using t-sne. J. Mach. Learn. Res. 9(11) (2008)"},{"issue":"3","key":"25_CR38","doi-asserted-by":"publisher","first-page":"325","DOI":"10.1109\/TAFFC.2017.2731763","volume":"10","author":"B Martinez","year":"2017","unstructured":"Martinez, B., Valstar, M.F., Jiang, B., Pantic, M.: Automatic analysis of facial actions: a survey. IEEE Trans. Affect. Comput. 10(3), 325\u2013347 (2017)","journal-title":"IEEE Trans. Affect. Comput."},{"issue":"2","key":"25_CR39","doi-asserted-by":"publisher","first-page":"151","DOI":"10.1109\/T-AFFC.2013.4","volume":"4","author":"SM Mavadati","year":"2013","unstructured":"Mavadati, S.M., Mahoor, M.H., Bartlett, K., Trinh, P., Cohn, J.F.: Disfa: a spontaneous facial action intensity database. IEEE Trans. Affect. Comput. 4(2), 151\u2013160 (2013)","journal-title":"IEEE Trans. Affect. Comput."},{"issue":"1","key":"25_CR40","doi-asserted-by":"publisher","first-page":"18","DOI":"10.1109\/TAFFC.2017.2740923","volume":"10","author":"A Mollahosseini","year":"2017","unstructured":"Mollahosseini, A., Hasani, B., Mahoor, M.H.: Affectnet: a database for facial expression, valence, and arousal computing in the wild. IEEE Trans. Affect. Comput. 10(1), 18\u201331 (2017)","journal-title":"IEEE Trans. Affect. Comput."},{"key":"25_CR41","doi-asserted-by":"crossref","unstructured":"Niu, X., Han, H., Yang, S., Huang, Y., Shan, S.: Local relationship learning with person-specific shape regularization for facial action unit detection. In: Proceedings of the IEEE\/CVF Conference on computer vision and pattern recognition, pp. 11917\u201311926 (2019)","DOI":"10.1109\/CVPR.2019.01219"},{"key":"25_CR42","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International conference on machine learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"25_CR43","doi-asserted-by":"crossref","unstructured":"Richardson, E., Alaluf, Y., Patashnik, O., Nitzan, Y., Azar, Y., Shapiro, S., Cohen-Or, D.: Encoding in style: a stylegan encoder for image-to-image translation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2287\u20132296 (2021)","DOI":"10.1109\/CVPR46437.2021.00232"},{"key":"25_CR44","doi-asserted-by":"crossref","unstructured":"Ridnik, T., et al.: Asymmetric loss for multi-label classification. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 82\u201391 (2021)","DOI":"10.1109\/ICCV48922.2021.00015"},{"key":"25_CR45","doi-asserted-by":"crossref","unstructured":"Selvaraju, R.R., Cogswell, M., Das, A., Vedantam, R., Parikh, D., Batra, D.: Grad-cam: Visual explanations from deep networks via gradient-based localization. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 618\u2013626 (2017)","DOI":"10.1109\/ICCV.2017.74"},{"key":"25_CR46","doi-asserted-by":"publisher","first-page":"321","DOI":"10.1007\/s11263-020-01378-z","volume":"129","author":"Z Shao","year":"2021","unstructured":"Shao, Z., Liu, Z., Cai, J., Ma, L.: JAA-net: joint facial action unit detection and face alignment via adaptive attention. Int. J. Comput. Vision 129, 321\u2013340 (2021)","journal-title":"Int. J. Comput. Vision"},{"key":"25_CR47","doi-asserted-by":"crossref","unstructured":"Song, T., Chen, L., Zheng, W., Ji, Q.: Uncertain graph neural networks for facial action unit detection. In: Proceedings of the AAAI Conference on Artificial Intelligence. vol.\u00a035, pp. 5993\u20136001 (2021)","DOI":"10.1609\/aaai.v35i7.16748"},{"key":"25_CR48","doi-asserted-by":"crossref","unstructured":"Song, T., Cui, Z., Zheng, W., Ji, Q.: Hybrid message passing with performance-driven structures for facial action unit detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6267\u20136276 (2021)","DOI":"10.1109\/CVPR46437.2021.00620"},{"key":"25_CR49","doi-asserted-by":"crossref","unstructured":"Tang, Y., Zeng, W., Zhao, D., Zhang, H.: Piap-df: pixel-interested and anti person-specific facial action unit detection net with discrete feedback learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 12899\u201312908 (2021)","DOI":"10.1109\/ICCV48922.2021.01266"},{"key":"25_CR50","doi-asserted-by":"crossref","unstructured":"Tu, C.H., Yang, C.Y., Hsu, J.Y.J.: Idennet: identity-aware facial action unit detection. In: 2019 14th IEEE International Conference on Automatic Face and Gesture Recognition (FG 2019), pp.\u00a01\u20138. IEEE (2019)","DOI":"10.1109\/FG.2019.8756631"},{"key":"25_CR51","unstructured":"Vaswani, A., et al.: Attention is all you need. Adv. Neural Inf. Proce. Syst. 30 (2017)"},{"key":"25_CR52","doi-asserted-by":"crossref","unstructured":"Wang, H., et al.: Dual-reference source-free active domain adaptation for nasopharyngeal carcinoma tumor segmentation across multiple hospitals. IEEE Trans. Med. Imaging (2024)","DOI":"10.1109\/TMI.2024.3412923"},{"key":"25_CR53","doi-asserted-by":"crossref","unstructured":"Wang, H., Jin, Y., Zhu, L.: Dynamic interactive relation capturing via scene graph learning for robotic surgical report generation. In: 2023 IEEE International Conference on Robotics and Automation (ICRA), pp. 2702\u20132709. IEEE (2023)","DOI":"10.1109\/ICRA48891.2023.10160647"},{"key":"25_CR54","doi-asserted-by":"crossref","unstructured":"Wang, H., et al.: Advancing uwf-slo vessel segmentation with source-free active domain adaptation and a novel multi-center dataset. arXiv preprint arXiv:2406.13645 (2024)","DOI":"10.1007\/978-3-031-72114-4_8"},{"key":"25_CR55","doi-asserted-by":"publisher","unstructured":"Wang, H., Zhang, S., Luo, X., Liao, W., Zhu, L.: Advancing delineation of gross tumor volume based on magnetic resonance imaging by performing source-free domain adaptation in nasopharyngeal carcinoma. In: International Workshop on Computational Mathematics Modeling in Cancer Analysis, pp. 71\u201380. Springer (2023). https:\/\/doi.org\/10.1007\/978-3-031-45087-7_8","DOI":"10.1007\/978-3-031-45087-7_8"},{"key":"25_CR56","unstructured":"Wang, H., et al.: Video-instrument synergistic network for referring video instrument segmentation in robotic surgery. arXiv preprint arXiv:2308.09475 (2023)"},{"key":"25_CR57","doi-asserted-by":"crossref","unstructured":"Wang, X., Chen, C.P., Yuan, H., Zhang, T.: Semantic learning for facial action unit detection. IEEE Trans. Comput. Soc. Syst. (2022)","DOI":"10.1109\/TCSS.2022.3166133"},{"key":"25_CR58","doi-asserted-by":"publisher","unstructured":"Wang, Y., See, J., Phan, R.C.W., Oh, Y.H.: LBP with six intersection points: reducing redundant information in LBP-top for micro-expression recognition. In: Computer Vision\u2013ACCV 2014: 12th Asian Conference on Computer Vision, Singapore, Singapore, November 1-5, 2014, Revised Selected Papers, Part I 12, pp. 525\u2013537. Springer (2015). https:\/\/doi.org\/10.1007\/978-3-319-16865-4_34","DOI":"10.1007\/978-3-319-16865-4_34"},{"key":"25_CR59","doi-asserted-by":"crossref","unstructured":"Wang, Z., Li, Y., Wang, S., Ji, Q.: Capturing global semantic relationships for facial action unit recognition. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 3304\u20133311 (2013)","DOI":"10.1109\/ICCV.2013.410"},{"key":"25_CR60","doi-asserted-by":"crossref","unstructured":"Wu, H., Yang, Y., Chen, H., Ren, J., Zhu, L.: Mask-guided progressive network for joint raindrop and rain streak removal in videos. In: Proceedings of the 31st ACM International Conference on Multimedia, pp. 7216\u20137225 (2023)","DOI":"10.1145\/3581783.3612001"},{"key":"25_CR61","doi-asserted-by":"crossref","unstructured":"Xing, Z., Ye, T., Yang, Y., Liu, G., Zhu, L.: Segmamba: Long-range sequential modeling mamba for 3d medical image segmentation. arXiv preprint arXiv:2401.13560 (2024)","DOI":"10.1007\/978-3-031-72111-3_54"},{"key":"25_CR62","doi-asserted-by":"publisher","unstructured":"Xing, Z., Yu, L., Wan, L., Han, T., Zhu, L.: Nestedformer: nested modality-aware transformer for brain tumor segmentation. In: International Conference on Medical Image Computing and Computer-Assisted Intervention. pp. 140\u2013150. Springer (2022). https:\/\/doi.org\/10.1007\/978-3-031-16443-9_14","DOI":"10.1007\/978-3-031-16443-9_14"},{"key":"25_CR63","unstructured":"Xing, Z., Zhu, L., Yu, L., Xing, Z., Wan, L.: c IEEE J. Biomed. Health Inf. (2024)"},{"key":"25_CR64","doi-asserted-by":"crossref","unstructured":"Yan, W.J., et al.: Casme ii: an improved spontaneous micro-expression database and the baseline evaluation. PLoS ONE 9(1), e86041 (2014)","DOI":"10.1371\/journal.pone.0086041"},{"key":"25_CR65","doi-asserted-by":"crossref","unstructured":"Yang, H., Yin, L., Zhou, Y., Gu, J.: Exploiting semantic embedding and visual feature for facial action unit detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10482\u201310491 (2021)","DOI":"10.1109\/CVPR46437.2021.01034"},{"key":"25_CR66","doi-asserted-by":"crossref","unstructured":"Yang, J., Shen, J., Lin, Y., Hristov, Y., Pantic, M.: Fan-trans: online knowledge distillation for facial action unit detection. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 6019\u20136027 (2023)","DOI":"10.1109\/WACV56688.2023.00596"},{"key":"25_CR67","doi-asserted-by":"crossref","unstructured":"Yin, Y., et al.: FG-net: Facial action unit detection with generalizable pyramidal features. arXiv preprint arXiv:2308.12380 (2023)","DOI":"10.1109\/WACV57701.2024.00599"},{"key":"25_CR68","doi-asserted-by":"crossref","unstructured":"Yin, Y., Lu, L., Wu, Y., Soleymani, M.: Self-supervised patch localization for cross-domain facial action unit detection. In: 2021 16th IEEE International Conference on Automatic Face and Gesture Recognition (FG 2021), pp.\u00a01\u20138. IEEE (2021)","DOI":"10.1109\/FG52635.2021.9667048"},{"key":"25_CR69","doi-asserted-by":"crossref","unstructured":"Zhang, L., Arandjelovic, O., Hong, X.: Facial action unit detection with local key facial sub-region based multi-label classification for micro-expression analysis. In: Proceedings of the 1st Workshop on Facial Micro-Expression: Advanced Techniques for Facial Expressions Generation and Spotting, pp. 11\u201318 (2021)","DOI":"10.1145\/3476100.3484462"},{"key":"25_CR70","doi-asserted-by":"crossref","unstructured":"Zhang, X., Yang, H., Wang, T., Li, X., Yin, L.: Multimodal channel-mixing: channel and spatial masked autoencoder on facial action unit detection. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 6077\u20136086 (2024)","DOI":"10.1109\/WACV57701.2024.00597"},{"key":"25_CR71","doi-asserted-by":"crossref","unstructured":"Zhang, X., et al.: Bp4d-spontaneous: a high-resolution spontaneous 3d dynamic facial expression database. Image Vis. Comput. 32(10), 692\u2013706 (2014)","DOI":"10.1016\/j.imavis.2014.06.002"},{"key":"25_CR72","unstructured":"Zhang, Y., Zhou, K., Liu, Z.: Neural prompt search. arXiv preprint arXiv:2206.04673 (2022)"},{"key":"25_CR73","unstructured":"Zhang, Y., Lu, H., Liu, X., Chen, Y., Wu, K.: Advancing generalizable remote physiological measurement through the integration of explicit and implicit prior knowledge. arXiv preprint arXiv:2403.06947 (2024)"},{"key":"25_CR74","doi-asserted-by":"crossref","unstructured":"Zhang, Z., et\u00a0al.: Multimodal spontaneous emotion corpus for human behavior analysis. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3438\u20133446 (2016)","DOI":"10.1109\/CVPR.2016.374"},{"issue":"6","key":"25_CR75","doi-asserted-by":"publisher","first-page":"915","DOI":"10.1109\/TPAMI.2007.1110","volume":"29","author":"G Zhao","year":"2007","unstructured":"Zhao, G., Pietikainen, M.: Dynamic texture recognition using local binary patterns with an application to facial expressions. IEEE Trans. Pattern Anal. Mach. Intell. 29(6), 915\u2013928 (2007)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"25_CR76","doi-asserted-by":"crossref","unstructured":"Zhao, K., Chu, W.S., De\u00a0la Torre, F., Cohn, J.F., Zhang, H.: Joint patch and multi-label learning for facial action unit detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2207\u20132216 (2015)","DOI":"10.1109\/CVPR.2015.7298833"},{"key":"25_CR77","doi-asserted-by":"crossref","unstructured":"Zhao, K., Chu, W.S., Zhang, H.: Deep region and multi-label learning for facial action unit detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3391\u20133399 (2016)","DOI":"10.1109\/CVPR.2016.369"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72973-7_25","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,15]],"date-time":"2025-02-15T15:00:53Z","timestamp":1739631653000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72973-7_25"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,1]]},"ISBN":["9783031729720","9783031729737"],"references-count":77,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72973-7_25","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,1]]},"assertion":[{"value":"1 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}