{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,29]],"date-time":"2026-07-29T06:04:32Z","timestamp":1785305072919,"version":"3.55.0"},"reference-count":90,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"the National Nature Science Foundation of China","doi-asserted-by":"crossref","award":["U22A2035"],"award-info":[{"award-number":["U22A2035"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"the National Nature Science Foundation of China","doi-asserted-by":"crossref","award":["U1736206"],"award-info":[{"award-number":["U1736206"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"the National Nature Science Foundation of China","doi-asserted-by":"crossref","award":["U1803262"],"award-info":[{"award-number":["U1803262"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"the National Social Science Fund of China","award":["19ZDA113"],"award-info":[{"award-number":["19ZDA113"]}]},{"name":"the Fundamental Research Funds for the Central Universities","award":["ZYTS25036"],"award-info":[{"award-number":["ZYTS25036"]}]},{"name":"Key Project of Hubei Provincial Health Commission","award":["WJ2023Z003"],"award-info":[{"award-number":["WJ2023Z003"]}]},{"name":"Major Project of Science and Technology Innovation of Hubei Province","award":["2024BCA003"],"award-info":[{"award-number":["2024BCA003"]}]},{"name":"Young Elite Scientists Sponsorship Program by CAST","award":["2023QNRC001"],"award-info":[{"award-number":["2023QNRC001"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1007\/s11263-026-02932-x","type":"journal-article","created":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T18:09:04Z","timestamp":1784052544000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Deception Detection Meets Vision-Language Models"],"prefix":"10.1007","volume":"134","author":[{"given":"Dongliang","family":"Zhu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5872-3872","authenticated-orcid":false,"given":"Ruimin","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mei","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiang","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Liang","family":"Liao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mang","family":"Ye","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,14]]},"reference":[{"issue":"5","key":"2932_CR1","doi-asserted-by":"publisher","first-page":"1042","DOI":"10.1109\/TIFS.2016.2639344","volume":"12","author":"M Abouelenien","year":"2016","unstructured":"Abouelenien, M., P\u00e9rez-Rosas, V., Mihalcea, R., & Burzo, M. (2016). Detecting deceptive behavior via integration of discriminative features from multiple modalities. IEEE Transactions on Information Forensics and Security, 12(5), 1042\u20131055.","journal-title":"IEEE Transactions on Information Forensics and Security"},{"key":"2932_CR2","unstructured":"Achiam, J., Adler, S., Agarwal, S., Ahmad, L., Akkaya, I., Aleman, F.L., Almeida, D., Altenschmidt, J., Altman, S., Anadkat, S., & McGrew, B. (2023). Gpt-4 technical report. arXiv:2303.08774"},{"key":"2932_CR3","doi-asserted-by":"publisher","first-page":"273","DOI":"10.1007\/s10579-007-9061-5","volume":"41","author":"J Allwood","year":"2007","unstructured":"Allwood, J., Cerrato, L., Jokinen, K., Navarretta, C., & Paggio, P. (2007). The mumin coding scheme for the annotation of feedback, turn management and sequencing phenomena. Language Resources and Evaluation, 41, 273\u2013287.","journal-title":"Language Resources and Evaluation"},{"key":"2932_CR4","doi-asserted-by":"crossref","unstructured":"Arnab, A., Dehghani, M., Heigold, G., Sun, C., Lu\u010di\u0107, M., & Schmid, C. (2021). Vivit: A video vision transformer. Proceedings of the IEEE\/CVF international conference on computer vision (pp. 6836\u20136846)","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"2932_CR5","doi-asserted-by":"crossref","unstructured":"Avola, D., Cinque, L., Foresti, G. L., & Pannone, D. (2019). Automatic deception detection in rgb videos using facial action units. Proceedings of the 13th International Conference on Distributed Smart Cameras (pp. 1\u20136)","DOI":"10.1145\/3349801.3349806"},{"key":"2932_CR6","doi-asserted-by":"crossref","unstructured":"Bai, C., Bolonkin, M., Burgoon, J., Chen, C., Dunbar, N., Singh, B., Subrahmanian, V., & Wu, Z. (2019). Automatic long-term deception detection in group interaction videos. 2019 IEEE International Conference on Multimedia and Expo (ICME) (pp. 1600\u20131605). IEEE.","DOI":"10.1109\/ICME.2019.00276"},{"key":"2932_CR7","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2023.110422","volume":"267","author":"N Bajaj","year":"2023","unstructured":"Bajaj, N., Rajwadi, M., Constance, T. G., Wall, J., Moniri, M., Laird, T., Woodruff, C., Laird, J., Glackin, C., & Cannings, N. (2023). Deception detection in conversations using the proximity of linguistic markers. Knowledge-Based Systems, 267, Article 110422.","journal-title":"Knowledge-Based Systems"},{"key":"2932_CR8","doi-asserted-by":"crossref","unstructured":"Baltrusaitis, T., Zadeh, A., Lim, Y.C., & Morency, L.P. (2018). Openface 2.0: Facial behavior analysis toolkit. In: 2018 13th IEEE International Conference on Automatic Face & Gesture Recognition (FG 2018), p\u00a059.","DOI":"10.1109\/FG.2018.00019"},{"issue":"2","key":"2932_CR9","doi-asserted-by":"publisher","first-page":"253","DOI":"10.1017\/S0048577299971664","volume":"36","author":"MS Bartlett","year":"1999","unstructured":"Bartlett, M. S., Hager, J. C., Ekman, P., & Sejnowski, T. J. (1999). Measuring facial expressions by computer image analysis. Psychophysiology, 36(2), 253\u2013263.","journal-title":"Psychophysiology"},{"key":"2932_CR10","unstructured":"Bertasius, G., Wang, H., & Torresani, L. (2021). Is space-time attention all you need for video understanding? In: ICML, vol\u00a02, p\u00a04."},{"issue":"3","key":"2932_CR11","doi-asserted-by":"publisher","first-page":"214","DOI":"10.1207\/s15327957pspr1003_2","volume":"10","author":"CF Bond Jr","year":"2006","unstructured":"Bond, C. F., Jr., & DePaulo, B. M. (2006). Accuracy of deception judgments. Personality and social psychology Review, 10(3), 214\u2013234.","journal-title":"Personality and social psychology Review"},{"key":"2932_CR12","doi-asserted-by":"crossref","unstructured":"Buch, S., Eyzaguirre, C., Gaidon, A., Wu, J., Fei-Fei, L., & Niebles, J. C. (2022). Revisiting the\" video\" in video-language understanding. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 2917\u20132927)","DOI":"10.1109\/CVPR52688.2022.00293"},{"key":"2932_CR13","doi-asserted-by":"crossref","unstructured":"Cai, R., Cui, Y., Yu, Z., Lin, X., Chen, C., & Kot, A. (2025). Rehearsal-free and efficient continual learning for cross-domain face anti-spoofing. IEEE Transactions on Pattern Analysis and Machine Intelligence,","DOI":"10.1109\/TPAMI.2025.3601053"},{"key":"2932_CR14","first-page":"342","volume":"7","author":"C Darwin","year":"1871","unstructured":"Darwin, C. (1871). The expression of the emotions in man and animals. History, 7, 342.","journal-title":"History"},{"key":"2932_CR15","doi-asserted-by":"crossref","unstructured":"DePaulo, B.M., & Morris, W.L. (2004). Discerning lies from truths: Behavioral cues to deception and the indirect pathway of intuition. BM DePaulo, WL Morris, ad by PA Granhag The detection of deception in forensic contexts\/\/New York: Cambridge pp 15\u201340.","DOI":"10.1017\/CBO9780511490071.002"},{"issue":"1","key":"2932_CR16","doi-asserted-by":"publisher","first-page":"74","DOI":"10.1037\/0033-2909.129.1.74","volume":"129","author":"BM DePaulo","year":"2003","unstructured":"DePaulo, B. M., Lindsay, J. J., Malone, B. E., Muhlenbruck, L., Charlton, K., & Cooper, H. (2003). Cues to deception. Psychological bulletin, 129(1), 74.","journal-title":"Psychological bulletin"},{"key":"2932_CR17","doi-asserted-by":"crossref","unstructured":"Ding, M., Zhao, A., Lu, Z., Xiang, T., & Wen, J. R. (2019). Face-focused cross-stream network for deception detection in videos. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 7802\u20137811)","DOI":"10.1109\/CVPR.2019.00799"},{"key":"2932_CR18","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s00521-024-09811-x","volume":"36","author":"L Dinges","year":"2024","unstructured":"Dinges, L., Fiedler, M. A., Al-Hamadi, A., Hempel, T., Abdelrahman, A., Weimann, J., Bershadskyy, D., & Steiner, J. (2024). Exploring facial cues: automated deception detection using artificial intelligence. Neural Computing and Applications, 36, 1\u201327. https:\/\/doi.org\/10.1007\/s00521-024-09811-x","journal-title":"Neural Computing and Applications"},{"key":"2932_CR19","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., & Gelly, S. (2020). An image is worth 16x16 words: Transformers for image recognition at scale arXiv:2010.11929"},{"issue":"6","key":"2932_CR20","doi-asserted-by":"publisher","first-page":"1441","DOI":"10.1037\/0022-3514.83.6.1441","volume":"83","author":"B Egloff","year":"2002","unstructured":"Egloff, B., & Schmukle, S. C. (2002). Predictive validity of an implicit association test for assessing anxiety. Journal of personality and social psychology, 83(6), 1441.","journal-title":"Journal of personality and social psychology"},{"key":"2932_CR21","unstructured":"Ekman, P. (2009). Telling lies: Clues to deceit in the marketplace, politics, and marriage (revised edition). WW Norton & Company"},{"issue":"1","key":"2932_CR22","doi-asserted-by":"publisher","first-page":"88","DOI":"10.1080\/00332747.1969.11023575","volume":"32","author":"P Ekman","year":"1969","unstructured":"Ekman, P., & Friesen, W. V. (1969). Nonverbal leakage and clues to deception. Psychiatry, 32(1), 88\u2013106.","journal-title":"Psychiatry"},{"key":"2932_CR23","doi-asserted-by":"crossref","unstructured":"Fang, H. S., Xie, S., Tai, Y. W., & Lu, C. (2017). Rmpe: Regional multi-person pose estimation. Proceedings of the IEEE international conference on computer vision (pp. 2334\u20132343)","DOI":"10.1109\/ICCV.2017.256"},{"key":"2932_CR24","doi-asserted-by":"publisher","first-page":"303","DOI":"10.1007\/s10506-013-9140-4","volume":"21","author":"T Fornaciari","year":"2013","unstructured":"Fornaciari, T., & Poesio, M. (2013). Automatic deception detection in italian court cases. Artificial intelligence and law, 21, 303\u2013340.","journal-title":"Artificial intelligence and law"},{"key":"2932_CR25","doi-asserted-by":"crossref","unstructured":"Foteinopoulou, N. M., & Patras, I. (2024). Emoclip: A vision-language method for zero-shot video facial expression recognition. 2024 IEEE 18th International Conference on Automatic Face and Gesture Recognition (FG) (pp. 1\u201310). IEEE.","DOI":"10.1109\/FG59268.2024.10581982"},{"key":"2932_CR26","doi-asserted-by":"crossref","unstructured":"Gogate, M., Adeel, A., & Hussain, A. (2017). Deep learning driven multimodal fusion for automated deception detection. 2017 IEEE symposium series on computational intelligence (pp. 1\u20136)","DOI":"10.1109\/SSCI.2017.8285382"},{"issue":"1","key":"2932_CR27","first-page":"83","volume":"7","author":"RB Graber","year":"1981","unstructured":"Graber, R. B. (1981). Ekman: The face of man: Expressions of universal emotions in a new guinea village. Stud Vis Commun, 7(1), 83\u201385.","journal-title":"Stud Vis Commun"},{"key":"2932_CR28","doi-asserted-by":"crossref","unstructured":"Guo, X., Selvaraj, N. M., Yu, Z., Kong, A., Shen, B., & Kot, A. (2023). Audio-visual deception detection: Dolos dataset and parameter-efficient crossmodal learning. Proceedings of the IEEE\/CVF International Conference on Computer Vision (pp. 22135\u201322145)","DOI":"10.1109\/ICCV51070.2023.02023"},{"key":"2932_CR29","doi-asserted-by":"crossref","unstructured":"Gupta, V., Agarwal, M., Arora, M., Chakraborty, T., Singh, R., & Vatsa, M. (2019). Bag-of-lies: A multimodal dataset for deception detection. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition workshops, pp 0\u20130","DOI":"10.1109\/CVPRW.2019.00016"},{"key":"2932_CR30","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 770\u2013778)","DOI":"10.1109\/CVPR.2016.90"},{"key":"2932_CR31","doi-asserted-by":"crossref","unstructured":"Hsiao, S. W., & Sun, C. Y. (2022). Attention-aware multi-modal rnn for deception detection. 2022 IEEE International Conference on Big Data (Big Data) (pp. 3593\u20133596)","DOI":"10.1109\/BigData55660.2022.10020331"},{"key":"2932_CR32","doi-asserted-by":"crossref","unstructured":"Jaiswal, M., Tabibu, S., & Bajpai, R. (2016). The truth and nothing but the truth: Multimodal analysis for deception detection. IEEE 16th International Conference on Data Mining Workshops (ICDMW) (pp. 938\u2013943). IEEE.","DOI":"10.1109\/ICDMW.2016.0137"},{"key":"2932_CR33","doi-asserted-by":"crossref","unstructured":"Jiang, B., Wang, M., Gan, W., Wu, W., & Yan, J. (2019). Stm: Spatiotemporal and motion encoding for action recognition. Proceedings of the IEEE\/CVF international conference on computer vision (pp. 2000\u20132009)","DOI":"10.1109\/ICCV.2019.00209"},{"key":"2932_CR34","doi-asserted-by":"crossref","unstructured":"Ju, C., Han, T., Zheng, K., Zhang, Y., & Xie, W. (2022). Prompting visual-language models for efficient video understanding. European Conference on Computer Vision (pp. 105\u2013124). Springer.","DOI":"10.1007\/978-3-031-19833-5_7"},{"key":"2932_CR35","doi-asserted-by":"crossref","unstructured":"Kang, J., Qu, W., Cui, S., & Feng, X. (2024). Deception detection algorithm based on global and local feature fusion with multi-head attention. 2024 3rd International Conference on Image Processing and Media Computing (ICIPMC) (pp. 162\u2013168). IEEE.","DOI":"10.1109\/ICIPMC62364.2024.10586666"},{"key":"2932_CR36","doi-asserted-by":"crossref","unstructured":"Karimi, H., Tang, J., & Li, Y. (2018). Toward end-to-end deception detection in videos. 2018 IEEE International Conference on Big Data (Big Data) (pp. 1278\u20131283)","DOI":"10.1109\/BigData.2018.8621909"},{"issue":"3","key":"2932_CR37","doi-asserted-by":"publisher","first-page":"971","DOI":"10.1109\/TCDS.2021.3086011","volume":"14","author":"M Karnati","year":"2021","unstructured":"Karnati, M., Seal, A., Yazidi, A., & Krejcar, O. (2021). Lienet: A deep convolution neural network framework for detecting deception. IEEE transactions on cognitive and developmental systems, 14(3), 971\u2013984.","journal-title":"IEEE transactions on cognitive and developmental systems"},{"key":"2932_CR38","doi-asserted-by":"crossref","unstructured":"Kim, M., Han, D., Kim, T., & Han, B. (2024). Leveraging temporal contextualization for video action recognition. In: European Conference on Computer Vision, Springer, pp 74\u201391","DOI":"10.1007\/978-3-031-72664-4_5"},{"issue":"7553","key":"2932_CR39","first-page":"436","volume":"521","author":"Y LeCun","year":"2015","unstructured":"LeCun, Y., Bengio, Y., & Hinton, G. (2015). Deep learning. nature, 521(7553), 436\u2013444.","journal-title":"Deep learning. nature"},{"key":"2932_CR40","unstructured":"Lei, J., Berg, T. L., & Bansal, M. (2022). Revealing single frame bias for video-and-language learning arXiv:2206.03428."},{"key":"2932_CR41","unstructured":"Li, B., Weinberger, K. Q., Belongie, S., Koltun, V., & Ranftl, R. (2022). Language-driven semantic segmentation arXiv:2201.03546."},{"key":"2932_CR42","doi-asserted-by":"crossref","unstructured":"Li, L., Zheng, Y., Liu, S., Xu, X., & Li, T. (2024). Domain knowledge enhanced vision-language pretrained model for dynamic facial expression recognition. In: Proceedings of the 32nd ACM International Conference on Multimedia, pp 5673\u20135682","DOI":"10.1145\/3664647.3681708"},{"key":"2932_CR43","doi-asserted-by":"crossref","unstructured":"Liang, F., Wu, B., Dai, X., Li, K., Zhao, Y., Zhang, H., Zhang, P., Vajda, P., & Marculescu, D. (2023). Open-vocabulary semantic segmentation with mask-adapted clip. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 7061\u20137070","DOI":"10.1109\/CVPR52729.2023.00682"},{"key":"2932_CR44","doi-asserted-by":"crossref","unstructured":"Lin, X., Liu, A., Yu, Z., Cai, R., Wang, S., Yu, Y., Wan, J., Lei, Z., Cao, X., & Kot, A. (2025). Reliable and balanced transfer learning for generalized multimodal face anti-spoofing. IEEE Transactions on Pattern Analysis and Machine Intelligence","DOI":"10.1109\/TPAMI.2025.3573785"},{"key":"2932_CR45","doi-asserted-by":"crossref","unstructured":"Liu, R., Huang, J., Li, G., Feng, J., Wu, X., & Li, T.H. (2023a). Revisiting temporal modeling for clip-based image-to-video knowledge transferring. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 6555\u20136564","DOI":"10.1109\/CVPR52729.2023.00634"},{"key":"2932_CR46","doi-asserted-by":"crossref","unstructured":"Liu, X., Park, D.H., Azadi, S., Zhang, G., Chopikyan, A., Hu, Y., Shi, H., Rohrbach, A., & Darrell, T. (2023b). More control for free! image synthesis with semantic diffusion guidance. In: Proceedings of the IEEE\/CVF winter conference on applications of computer vision, pp 289\u2013299","DOI":"10.1109\/WACV56688.2023.00037"},{"key":"2932_CR47","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., & Guo, B. (2021). Swin transformer: Hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 10012\u201310022","DOI":"10.1109\/ICCV48922.2021.00986"},{"issue":"1","key":"2932_CR48","doi-asserted-by":"publisher","first-page":"429","DOI":"10.3758\/s13428-018-1061-4","volume":"51","author":"EP Lloyd","year":"2019","unstructured":"Lloyd, E. P., Deska, J. C., Hugenberg, K., McConnell, A. R., Humphrey, B. T., & Kunstman, J. W. (2019). Miami university deception detection database. Behavior research methods, 51(1), 429\u2013439.","journal-title":"Behavior research methods"},{"key":"2932_CR49","doi-asserted-by":"publisher","first-page":"15968","DOI":"10.1609\/aaai.v35i18.17980","volume":"35","author":"L Mathur","year":"2021","unstructured":"Mathur, L. (2021). Affect-aware machine learning models for deception detection. Proceedings of the AAAI Conference on Artificial Intelligence, 35, 15968\u201315969.","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"2932_CR50","doi-asserted-by":"crossref","unstructured":"Mathur, L., & Matari\u0107, M. J. (2020). Introducing representations of facial affect in automated multimodal deception detection. Proceedings of the 2020 international conference on multimodal interaction (pp. 305\u2013314)","DOI":"10.1145\/3382507.3418864"},{"key":"2932_CR51","doi-asserted-by":"publisher","DOI":"10.3389\/fpsyg.2018.02545","volume":"9","author":"D Matsumoto","year":"2018","unstructured":"Matsumoto, D., & Hwang, H. C. (2018). Microexpressions differentiate truths from lies about future malicious intent. Frontiers in psychology, 9, Article 408238.","journal-title":"Frontiers in psychology"},{"key":"2932_CR52","doi-asserted-by":"crossref","unstructured":"Momeni, L., Caron, M., Nagrani, A., Zisserman, A., & Schmid, C. (2023). Verbs in action: Improving verb understanding in video-language models. Proceedings of the IEEE\/CVF International Conference on Computer Vision (pp. 15579\u201315591)","DOI":"10.1109\/ICCV51070.2023.01428"},{"key":"2932_CR53","doi-asserted-by":"publisher","DOI":"10.1016\/j.chb.2021.107063","volume":"127","author":"M Monaro","year":"2022","unstructured":"Monaro, M., Maldera, S., Scarpazza, C., Sartori, G., & Navarin, N. (2022). Detecting deception through facial expressions in a dataset of videotaped interviews: A comparison between human judges and machine learning models. Computers in Human Behavior, 127, Article 107063.","journal-title":"Computers in Human Behavior"},{"issue":"22","key":"2932_CR54","doi-asserted-by":"publisher","first-page":"27413","DOI":"10.1007\/s10489-023-04968-9","volume":"53","author":"B Nam","year":"2023","unstructured":"Nam, B., Kim, J. Y., Bark, B., Kim, Y., Kim, J., So, S. W., Choi, H. Y., & Kim, I. Y. (2023). Facialcuenet: unmasking deception-an interpretable model for criminal interrogation using facial expressions. Applied Intelligence, 53(22), 27413\u201327427.","journal-title":"Applied Intelligence"},{"key":"2932_CR55","doi-asserted-by":"crossref","unstructured":"Ni, B., Peng, H., Chen, M., Zhang, S., Meng, G., Fu, J., Xiang, S., & Ling, H. (2022). Expanding language-image pretrained models for general video recognition. European conference on computer vision (pp. 1\u201318). Springer.","DOI":"10.1007\/978-3-031-19772-7_1"},{"key":"2932_CR56","doi-asserted-by":"publisher","first-page":"26462","DOI":"10.52202\/068431-1919","volume":"35","author":"J Pan","year":"2022","unstructured":"Pan, J., Lin, Z., Zhu, X., Shao, J., & Li, H. (2022). St-adapter: Parameter-efficient image-to-video transfer learning. Advances in Neural Information Processing Systems, 35, 26462\u201326477.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2932_CR57","doi-asserted-by":"crossref","unstructured":"P\u00e9rez-Rosas, V., Abouelenien, M., Mihalcea, R., & Burzo, M. (2015a). Deception detection using real-life trial data. In: Proceedings of the 2015 ACM on international conference on multimodal interaction, pp 59\u201366","DOI":"10.1145\/2818346.2820758"},{"key":"2932_CR58","doi-asserted-by":"crossref","unstructured":"P\u00e9rez-Rosas, V., Abouelenien, M., Mihalcea, R., Xiao, Y., Linton, C., & Burzo, M. (2015b). Verbal and nonverbal clues for real-life deception detection. In: Proceedings of the 2015 conference on empirical methods in natural language processing, pp 2336\u20132346","DOI":"10.18653\/v1\/D15-1281"},{"key":"2932_CR59","doi-asserted-by":"crossref","unstructured":"Qing, Z., Zhang, S., Huang, Z., Zhang, Y., Gao, C., Zhao, D., & Sang, N. (2023). Disentangling spatial and temporal learning for efficient image-to-video transfer learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 13934\u201313944","DOI":"10.1109\/ICCV51070.2023.01281"},{"key":"2932_CR60","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., & et\u00a0al. (2021). Learning transferable visual models from natural language supervision. International conference on machine learning (pp. 8748\u20138763). PmLR"},{"key":"2932_CR61","doi-asserted-by":"crossref","unstructured":"Rao, Y., Zhao, W., Chen, G., Tang, Y., Zhu, Z., Huang, G., Zhou, J., & Lu, J. (2022). Denseclip: Language-guided dense prediction with context-aware prompting. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 18082\u201318091)","DOI":"10.1109\/CVPR52688.2022.01755"},{"key":"2932_CR62","doi-asserted-by":"crossref","unstructured":"Redmon, J., Divvala, S., Girshick, R., & Farhadi, A. (2016). You only look once: Unified, real-time object detection. Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 779\u2013788)","DOI":"10.1109\/CVPR.2016.91"},{"key":"2932_CR63","doi-asserted-by":"crossref","unstructured":"Rill-Garcia, R., Jair\u00a0Escalante, H., Villasenor-Pineda, L., & Reyes-Meza, V. (2019). High-level features for multimodal deception detection in videos. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition workshops, pp 0\u20130","DOI":"10.52591\/lxai201906152"},{"issue":"1","key":"2932_CR64","doi-asserted-by":"publisher","first-page":"306","DOI":"10.1109\/TAFFC.2020.3015684","volume":"13","author":"MU \u015een","year":"2020","unstructured":"\u015een, M. U., Perez-Rosas, V., Yanikoglu, B., Abouelenien, M., Burzo, M., & Mihalcea, R. (2020). Multimodal deception detection using real-life trial data. IEEE Transactions on Affective Computing, 13(1), 306\u2013319.","journal-title":"IEEE Transactions on Affective Computing"},{"key":"2932_CR65","doi-asserted-by":"crossref","unstructured":"Sevilla-Lara, L., Zha, S., Yan, Z., Goswami, V., Feiszli, M., & Torresani, L. (2021). Only time can tell: Discovering temporal data for temporal modeling. Proceedings of the IEEE\/CVF winter conference on applications of computer vision (pp. 535\u2013544)","DOI":"10.1109\/WACV48630.2021.00058"},{"key":"2932_CR66","doi-asserted-by":"crossref","unstructured":"Tao, M., Bao, B. K., Tang, H., & Xu, C. (2023). Galip: Generative adversarial clips for text-to-image synthesis. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 14214\u201314223)","DOI":"10.1109\/CVPR52729.2023.01366"},{"key":"2932_CR67","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., & Polosukhin, I. (2017). Attention is all you need. Advances in neural information processing systems 30"},{"key":"2932_CR68","doi-asserted-by":"crossref","unstructured":"Venkatesh, S., Ramachandra, R., Bours, P. (2020). Video based deception detection using deep recurrent convolutional neural network. In: Computer Vision and Image Processing: 4th International Conference, CVIP 2019, Jaipur, India, September 27\u201329, 2019, Revised Selected Papers, Part II 4, Springer, pp 163\u2013169","DOI":"10.1007\/978-981-15-4018-9_15"},{"key":"2932_CR69","doi-asserted-by":"crossref","unstructured":"Wang, L., Xiong, Y., Wang, Z., Qiao, Y., Lin, D., Tang, X., Van\u00a0Gool, L. (2016). Temporal segment networks: Towards good practices for deep action recognition. In: European conference on computer vision, Springer, pp 20\u201336","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"2932_CR70","doi-asserted-by":"crossref","unstructured":"Wang, L., Tong, Z., Ji, B., & Wu, G. (2021a). Tdn: Temporal difference networks for efficient action recognition. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 1895\u20131904","DOI":"10.1109\/CVPR46437.2021.00193"},{"key":"2932_CR71","unstructured":"Wang, M., Xing, J., Liu, Y.(2021b). Actionclip: A new paradigm for video action recognition. arXiv:2109.08472"},{"issue":"3","key":"2932_CR72","first-page":"3347","volume":"45","author":"M Wang","year":"2022","unstructured":"Wang, M., Xing, J., Su, J., Chen, J., & Liu, Y. (2022). Learning spatiotemporal and motion features in a unified 2d network for action recognition. IEEE Transactions on Pattern Analysis and Machine Intelligence, 45(3), 3347\u20133362.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2932_CR73","doi-asserted-by":"crossref","unstructured":"Wasim, S.T., Naseer, M., Khan, S., Khan, F.S., & Shah, M. (2023). Vita-clip: Video and text adaptive clip via multimodal prompting. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 23034\u201323044","DOI":"10.1109\/CVPR52729.2023.02206"},{"key":"2932_CR74","doi-asserted-by":"publisher","first-page":"2847","DOI":"10.1609\/aaai.v37i3.25386","volume":"37","author":"W Wu","year":"2023","unstructured":"Wu, W., Sun, Z., & Ouyang, W. (2023). Revisiting classifier: Transferring vision-language models for video recognition. Proceedings of the AAAI conference on artificial intelligence, 37, 2847\u20132855.","journal-title":"Proceedings of the AAAI conference on artificial intelligence"},{"key":"2932_CR75","doi-asserted-by":"crossref","unstructured":"Wu, W., Wang, X., Luo, H., Wang, J., Yang, Y., & Ouyang, W. (2023b). Bidirectional cross-modal knowledge exploration for video recognition with pre-trained vision-language models. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 6620\u20136630","DOI":"10.1109\/CVPR52729.2023.00640"},{"key":"2932_CR76","doi-asserted-by":"crossref","unstructured":"Wu, Z., Singh, B., Davis, L., & Subrahmanian, V. (2018). Deception detection in videos. Proceedings of the AAAI conference on artificial intelligence (Vol. 32,","DOI":"10.1609\/aaai.v32i1.11502"},{"issue":"1","key":"2932_CR77","doi-asserted-by":"publisher","first-page":"37","DOI":"10.1007\/s44267-024-00072-9","volume":"2","author":"X Xie","year":"2024","unstructured":"Xie, X., Cui, Y., Tan, T., Zheng, X., & Yu, Z. (2024). Fusionmamba: Dynamic feature enhancement for multimodal image fusion with mamba. Visual Intelligence, 2(1), 37.","journal-title":"Visual Intelligence"},{"key":"2932_CR78","unstructured":"Yang, T., Zhu, Y., Xie, Y., Zhang, A., Chen, C., & Li, M. (2023). Aim: Adapting image models for efficient video action recognition arXiv:2302.03024."},{"key":"2932_CR79","doi-asserted-by":"crossref","unstructured":"Yap, M.H., Ugail, H., Zwiggelaar, R. (2013). A database for facial behavioural analysis. In: 2013 10th IEEE International Conference and Workshops on Automatic Face and Gesture Recognition (FG), IEEE, pp 1\u20136","DOI":"10.1109\/FG.2013.6553803"},{"key":"2932_CR80","doi-asserted-by":"crossref","unstructured":"Ye, Q., Yu, Z., Shao, R., Cui, Y., Kang, X., Liu, X., Torr, P., & Cao, X. (2025). Cat+: Investigating and enhancing audio-visual understanding in large language models. IEEE Transactions on Pattern Analysis and Machine Intelligence","DOI":"10.1109\/TPAMI.2025.3582389"},{"issue":"6","key":"2932_CR81","doi-asserted-by":"publisher","first-page":"1307","DOI":"10.1007\/s11263-023-01758-1","volume":"131","author":"Z Yu","year":"2023","unstructured":"Yu, Z., Shen, Y., Shi, J., Zhao, H., Cui, Y., Zhang, J., Torr, P., & Zhao, G. (2023). Physformer++: Facial video-based physiological measurement with slowfast temporal difference transformer. International Journal of Computer Vision, 131(6), 1307\u20131330.","journal-title":"International Journal of Computer Vision"},{"key":"2932_CR82","doi-asserted-by":"crossref","unstructured":"Yu, Z., Cai, R., Cui, Y., Liu, X., Hu, Y., & Kot, A. C. (2024). Rethinking vision transformer and masked autoencoder in multimodal face anti-spoofing. International Journal of Computer Vision,132(11), 5217\u20135238.","DOI":"10.1007\/s11263-024-02055-1"},{"key":"2932_CR83","doi-asserted-by":"crossref","unstructured":"Zhai, X., Kolesnikov, A., Houlsby, N., & Beyer, L. (2022). Scaling vision transformers. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 12104\u201312113)","DOI":"10.1109\/CVPR52688.2022.01179"},{"issue":"2","key":"2932_CR84","doi-asserted-by":"publisher","first-page":"284","DOI":"10.1007\/s11633-025-1625-x","volume":"23","author":"J Zhang","year":"2026","unstructured":"Zhang, J., Lin, X., Huang, J., Ye, S., Guo, X., Zhu, D., Hu, R., Guo, D., Liang, Y., Yu, Z., et al. (2026). Multimodal deception detection: A survey. Machine Intelligence Research, 23(2), 284\u2013307.","journal-title":"Machine Intelligence Research"},{"key":"2932_CR85","unstructured":"Zhao, Z., & Patras, I. (2023). Prompting visual-language models for dynamic facial expression recognition arXiv:2308.13382."},{"key":"2932_CR86","doi-asserted-by":"crossref","unstructured":"Zhou, K., Yang, J., Loy, C.C., & Liu, Z. (2022a). Conditional prompt learning for vision-language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp 16816\u201316825","DOI":"10.1109\/CVPR52688.2022.01631"},{"issue":"9","key":"2932_CR87","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou, K., Yang, J., Loy, C. C., & Liu, Z. (2022). Learning to prompt for vision-language models. International Journal of Computer Vision, 130(9), 2337\u20132348.","journal-title":"International Journal of Computer Vision"},{"key":"2932_CR88","doi-asserted-by":"crossref","unstructured":"Zhou, Z., Lei, Y., Zhang, B., Liu, L., & Liu, Y. (2023). Zegclip: Towards adapting clip for zero-shot semantic segmentation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 11175\u201311185","DOI":"10.1109\/CVPR52729.2023.01075"},{"key":"2932_CR89","unstructured":"Zhu, D., Niu, Z., Zhao, B., Huang, J., Ye, S., Lin, X., Ma, H., Wang, T., Zhang, J., Zhu, C., Cao, J., Ma, Y., Song, R., Clap\u00e9s, A., Escalera, S., Guo, D., & Yu, Z. (2026). Svc 2026: the second multimodal deception detection challenge and the first domain generalized remote physiological measurement challenge., arXiv: 2604.05748"},{"key":"2932_CR90","doi-asserted-by":"crossref","unstructured":"Zhuo, Y., Baskaran, V. M., Kiaw, L., & Phan, R. (2024). Video deception detection through the fusion of multimodal feature extraction and neural networks. 2024 International Joint Conference on Neural Networks (IJCNN) (pp. 1\u20138). IEEE.","DOI":"10.1109\/IJCNN60899.2024.10651008"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02932-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-026-02932-x","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02932-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,29]],"date-time":"2026-07-29T05:50:40Z","timestamp":1785304240000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-026-02932-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":90,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2026,7]]}},"alternative-id":["2932"],"URL":"https:\/\/doi.org\/10.1007\/s11263-026-02932-x","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"22 July 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 June 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 July 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"349"}}