{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,9]],"date-time":"2026-04-09T00:41:19Z","timestamp":1775695279067,"version":"3.50.1"},"reference-count":101,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2025,7,29]],"date-time":"2025-07-29T00:00:00Z","timestamp":1753747200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,7,29]],"date-time":"2025-07-29T00:00:00Z","timestamp":1753747200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"the Natural Science Foundation of China","doi-asserted-by":"crossref","award":["61602345, 62002263"],"award-info":[{"award-number":["61602345, 62002263"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"the Natural Science Foundation of China","doi-asserted-by":"crossref","award":["61602345, 62002263"],"award-info":[{"award-number":["61602345, 62002263"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"the Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62272144"],"award-info":[{"award-number":["62272144"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"the National Key R&D Program of China","award":["2019YFB2101900"],"award-info":[{"award-number":["2019YFB2101900"]}]},{"name":"the National Key R&D Program of China","award":["2019YFB2101900"],"award-info":[{"award-number":["2019YFB2101900"]}]},{"name":"the National Key R&D Program of China","award":["2024YFB3311602"],"award-info":[{"award-number":["2024YFB3311602"]}]},{"name":"the TianKai Higher Education Innovation Park Enterprise R&D Special Project","award":["23YFZXYC00046"],"award-info":[{"award-number":["23YFZXYC00046"]}]},{"name":"the TianKai Higher Education Innovation Park Enterprise R&D Special Project","award":["23YFZXYC00046"],"award-info":[{"award-number":["23YFZXYC00046"]}]},{"name":"the Major Project of Anhui Province","award":["2408085J040"],"award-info":[{"award-number":["2408085J040"]}]},{"name":"the Fundamental Research Funds for the Central Universities","award":["JZ2024HGTG0309, JZ2024AHST0337"],"award-info":[{"award-number":["JZ2024HGTG0309, JZ2024AHST0337"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Machine Vision and Applications"],"published-print":{"date-parts":[[2025,9]]},"DOI":"10.1007\/s00138-025-01720-w","type":"journal-article","created":{"date-parts":[[2025,7,29]],"date-time":"2025-07-29T14:07:47Z","timestamp":1753798067000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["MSPhys: multiscale fusing-based diffusion model for remote physiological measurement"],"prefix":"10.1007","volume":"36","author":[{"given":"Gaoji","family":"Su","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Qian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qi","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"YingXu","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weijia","family":"Feng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dan","family":"Guo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,7,29]]},"reference":[{"issue":"3","key":"1720_CR1","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1088\/0967-3334\/28\/3\/R01","volume":"28","author":"J Allen","year":"2007","unstructured":"Allen, J.: Photoplethysmography and its application in clinical physiological measurement. Physiol. Meas. 28(3), 1 (2007)","journal-title":"Physiol. Meas."},{"key":"1720_CR2","doi-asserted-by":"publisher","DOI":"10.1016\/j.cosrev.2021.100399","volume":"40","author":"R Arya","year":"2021","unstructured":"Arya, R., Singh, J., Kumar, A.: A survey of multidisciplinary domains contributing to affective computing. Comput. Sci. Rev. 40, 100399 (2021)","journal-title":"Comput. Sci. Rev."},{"issue":"1","key":"1720_CR3","doi-asserted-by":"publisher","first-page":"72","DOI":"10.1038\/s41746-019-0150-9","volume":"2","author":"DR Seshadri","year":"2019","unstructured":"Seshadri, D.R., Li, R.T., Voos, J.E., Rowbottom, J.R., Alfes, C.M., Zorman, C.A., Drummond, C.K.: Wearable sensors for monitoring the physiological and biochemical profile of the athlete. NPJ Digital Med. 2(1), 72 (2019)","journal-title":"NPJ Digital Med."},{"issue":"6","key":"1720_CR4","doi-asserted-by":"publisher","first-page":"1358","DOI":"10.1109\/TCSS.2020.3033302","volume":"7","author":"H Liu","year":"2020","unstructured":"Liu, H., Chatterjee, I., Zhou, M., Lu, X.S., Abusorrah, A.: Aspect-based sentiment analysis: a survey of deep learning methods. IEEE Trans. Comput. Soc. Syst. 7(6), 1358\u20131375 (2020)","journal-title":"IEEE Trans. Comput. Soc. Syst."},{"issue":"8","key":"1720_CR5","doi-asserted-by":"publisher","first-page":"2781","DOI":"10.1109\/TCSVT.2019.2926632","volume":"30","author":"J Shi","year":"2020","unstructured":"Shi, J., Alikhani, I., Li, X., Yu, Z., Sepp\u00e4nen, T., Zhao, G.: Atrial fibrillation detection from face videos by fusing subtle variations. IEEE Trans. Circuits Syst. Video Technol. 30(8), 2781\u20132795 (2020)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"1720_CR6","doi-asserted-by":"crossref","unstructured":"Liu, Y., Jourabloo, A., Liu, X.: Learning deep models for face anti-spoofing: binary or auxiliary supervision. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 389\u2013398 (2018)","DOI":"10.1109\/CVPR.2018.00048"},{"key":"1720_CR7","doi-asserted-by":"crossref","unstructured":"Huang, Z., Wang, W., Haan, G.d.: Nose breathing or mouth breathing? A thermography-based new measurement for sleep monitoring. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 3882\u20133888 (2021)","DOI":"10.1109\/CVPRW53098.2021.00430"},{"issue":"4","key":"1720_CR8","doi-asserted-by":"publisher","first-page":"3305","DOI":"10.1109\/TAFFC.2023.3238641","volume":"14","author":"C\u00c1 Casado","year":"2023","unstructured":"Casado, C.\u00c1., Ca\u00f1ellas, M.L., L\u00f3pez, M.B.: Depression recognition using remote photoplethysmography from facial videos. IEEE Trans. Affect. Comput. 14(4), 3305\u20133316 (2023)","journal-title":"IEEE Trans. Affect. Comput."},{"issue":"1","key":"1720_CR9","doi-asserted-by":"publisher","first-page":"010901","DOI":"10.1117\/1.JBO.29.1.010901","volume":"29","author":"K Setchfield","year":"2024","unstructured":"Setchfield, K., Gorman, A., Simpson, A., Somekh, M.G., Wright, A.J.: Effect of skin color on optical properties and the implications for medical optical technologies: a review. J. Biomed. Opt. 29(1), 010901\u2013010901 (2024)","journal-title":"J. Biomed. Opt."},{"key":"1720_CR10","doi-asserted-by":"crossref","unstructured":"Speth, J., Vance, N., Flynn, P., Bowyer, K., Czajka, A.: Remote pulse estimation in the presence of face masks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition pp. 2086\u20132095 (2022)","DOI":"10.1109\/CVPRW56347.2022.00226"},{"key":"1720_CR11","doi-asserted-by":"crossref","unstructured":"Lam, A., Kuno, Y.: Robust heart rate measurement from video using select random patches. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3640\u20133648 (2015)","DOI":"10.1109\/ICCV.2015.415"},{"issue":"1","key":"1720_CR12","doi-asserted-by":"publisher","first-page":"7","DOI":"10.1109\/TBME.2010.2086456","volume":"58","author":"M-Z Poh","year":"2010","unstructured":"Poh, M.-Z., McDuff, D.J., Picard, R.W.: Advancements in noncontact, multiparameter physiological measurements using a webcam. IEEE Trans. Biomed. Eng. 58(1), 7\u201311 (2010)","journal-title":"IEEE Trans. Biomed. Eng."},{"key":"1720_CR13","unstructured":"Lewandowska, M., Rumi\u0144ski, J., Kocejko, T., Nowak, J.: Measuring pulse rate with a webcam\u2013a non-contact method for evaluating cardiac activity. In: Proc. FedCSIS, pp. 405\u2013410 (2011)"},{"issue":"10","key":"1720_CR14","doi-asserted-by":"publisher","first-page":"2878","DOI":"10.1109\/TBME.2013.2266196","volume":"60","author":"G De Haan","year":"2013","unstructured":"De Haan, G., Jeanne, V.: Robust pulse rate from chrominance-based RPPG. IEEE Trans. Biomed. Eng. 60(10), 2878\u20132886 (2013)","journal-title":"IEEE Trans. Biomed. Eng."},{"issue":"9","key":"1720_CR15","doi-asserted-by":"publisher","first-page":"1974","DOI":"10.1109\/TBME.2015.2508602","volume":"63","author":"W Wang","year":"2015","unstructured":"Wang, W., Stuijk, S., De Haan, G.: A novel algorithm for remote photoplethysmography: spatial subspace rotation. IEEE Trans. Biomed. Eng. 63(9), 1974\u20131984 (2015)","journal-title":"IEEE Trans. Biomed. Eng."},{"issue":"10","key":"1720_CR16","doi-asserted-by":"publisher","first-page":"7411","DOI":"10.1109\/TIM.2020.2984168","volume":"69","author":"R Song","year":"2020","unstructured":"Song, R., Zhang, S., Li, C., Zhang, Y., Cheng, J., Chen, X.: Heart rate estimation from facial videos using a spatiotemporal representation with convolutional neural networks. IEEE Trans. Instrum. Meas. 69(10), 7411\u20137421 (2020)","journal-title":"IEEE Trans. Instrum. Meas."},{"key":"1720_CR17","doi-asserted-by":"publisher","first-page":"2409","DOI":"10.1109\/TIP.2019.2947204","volume":"29","author":"X Niu","year":"2019","unstructured":"Niu, X., Shan, S., Han, H., Chen, X.: Rhythmnet: end-to-end heart rate estimation from face via spatial-temporal representation. IEEE Trans. Image Process. 29, 2409\u20132423 (2019)","journal-title":"IEEE Trans. Image Process."},{"key":"1720_CR18","doi-asserted-by":"crossref","unstructured":"Niu, X., Yu, Z., Han, H., Li, X., Shan, S., Zhao, G.: Video-based remote physiological measurement via cross-verified feature disentangling. In: Proceedings of the European Conference on Computer Vision, pp. 295\u2013310 (2020)","DOI":"10.1007\/978-3-030-58536-5_18"},{"key":"1720_CR19","first-page":"19400","volume":"33","author":"X Liu","year":"2020","unstructured":"Liu, X., Fromm, J., Patel, S., McDuff, D.: Multi-task temporal shift attention networks for on-device contactless vitals measurement. Adv. Neural. Inf. Process. Syst. 33, 19400\u201319411 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1720_CR20","doi-asserted-by":"crossref","unstructured":"Liu, X., Hill, B., Jiang, Z., Patel, S., McDuff, D.: Efficientphys: enabling simple, fast and accurate camera-based cardiac measurement. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 5008\u20135017 (2023)","DOI":"10.1109\/WACV56688.2023.00498"},{"key":"1720_CR21","unstructured":"\u0160petl\u00edk, R., Franc, V., Matas, J.: Visual heart rate estimation with convolutional neural network. In: Proceedings of the British Machine Vision Conference, pp. 3\u20136 (2018)"},{"issue":"20","key":"1720_CR22","doi-asserted-by":"publisher","first-page":"4364","DOI":"10.3390\/app9204364","volume":"9","author":"F Bousefsaf","year":"2019","unstructured":"Bousefsaf, F., Pruski, A., Maaoui, C.: 3d convolutional neural networks for remote pulse rate measurement and mapping from facial video. Appl. Sci. 9(20), 4364 (2019)","journal-title":"Appl. Sci."},{"key":"1720_CR23","doi-asserted-by":"crossref","unstructured":"Yu, Z., Peng, W., Li, X., Hong, X., Zhao, G.: Remote heart rate measurement from highly compressed facial videos: an end-to-end deep learning solution with video enhancement. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 151\u2013160 (2019)","DOI":"10.1109\/ICCV.2019.00024"},{"key":"1720_CR24","doi-asserted-by":"crossref","unstructured":"Chen, W., McDuff, D.: Deepphys: video-based physiological measurement using convolutional attention networks. In: Proceedings of the European Conference on Computer Vision, pp. 349\u2013365 (2018)","DOI":"10.1007\/978-3-030-01216-8_22"},{"key":"1720_CR25","doi-asserted-by":"crossref","unstructured":"Li, K., Wang, Y., Zhang, J., Gao, P., Song, G., Liu, Y., Li, H., Qiao, Y.: Uniformer: unifying convolution and self-attention for visual recognition. arXiv:2201.09450 (2022)","DOI":"10.1109\/TPAMI.2023.3282631"},{"key":"1720_CR26","doi-asserted-by":"crossref","unstructured":"Perepelkina, O., Artemyev, M., Churikova, M., Grinenko, M.: Hearttrack: convolutional neural network for remote video-based heart rate monitoring. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops, pp. 288\u2013289 (2020)","DOI":"10.1109\/CVPRW50498.2020.00152"},{"key":"1720_CR27","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo, B.: Swin transformer: hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10012\u201310022 (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"1720_CR28","doi-asserted-by":"crossref","unstructured":"Qian, W., Su, G., Guo, D., Zhou, J., Li, X., Hu, B., Tang, S., Wang, M.: Physdiff: physiology-based dynamicity disentangled diffusion model for remote physiological measurement. In: Proceedings of the AAAI Conference on Artificial Intelligence (AAAI) (2025)","DOI":"10.1609\/aaai.v39i6.32704"},{"issue":"6","key":"1720_CR29","doi-asserted-by":"publisher","first-page":"1307","DOI":"10.1007\/s11263-023-01758-1","volume":"131","author":"Z Yu","year":"2023","unstructured":"Yu, Z., Shen, Y., Shi, J., Zhao, H., Cui, Y., Zhang, J., Torr, P., Zhao, G.: Physformer++: facial video-based physiological measurement with slowfast temporal difference transformer. Int. J. Comput. Vis. 131(6), 1307\u20131330 (2023)","journal-title":"Int. J. Comput. Vis."},{"key":"1720_CR30","doi-asserted-by":"crossref","unstructured":"Yu, Z., Shen, Y., Shi, J., Zhao, H., Torr, P., Zhao, G.: Physformer: facial video-based physiological measurement with temporal difference transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4186\u20134196 (2022)","DOI":"10.1109\/CVPR52688.2022.00415"},{"key":"1720_CR31","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1186\/s12938-017-0395-y","volume":"16","author":"A Al-Naji","year":"2017","unstructured":"Al-Naji, A., Perera, A.G., Chahl, J.: Remote monitoring of cardiorespiratory signals from a hovering unmanned aerial vehicle. Biomed. Eng. Online 16, 1\u201320 (2017)","journal-title":"Biomed. Eng. Online"},{"key":"1720_CR32","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1186\/s12938-016-0300-0","volume":"16","author":"B Wei","year":"2017","unstructured":"Wei, B., He, X., Zhang, C., Wu, X.: Non-contact, synchronous dynamic measurement of respiratory rate and heart rate based on dual sensitive regions. Biomed. Eng. Online 16, 1\u201321 (2017)","journal-title":"Biomed. Eng. Online"},{"key":"1720_CR33","doi-asserted-by":"crossref","unstructured":"Guo, Z., Wang, Z.J., Shen, Z.: Physiological parameter monitoring of drivers based on video data and independent vector analysis. In: 2014 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4374\u20134378 (2014). IEEE","DOI":"10.1109\/ICASSP.2014.6854428"},{"issue":"2","key":"1720_CR34","doi-asserted-by":"publisher","first-page":"243","DOI":"10.3390\/bioengineering10020243","volume":"10","author":"F Haugg","year":"2023","unstructured":"Haugg, F., Elgendi, M., Menon, C.: GRGB rPPG: an efficient low-complexity remote photoplethysmography-based algorithm for heart rate estimation. Bioengineering 10(2), 243 (2023)","journal-title":"Bioengineering"},{"key":"1720_CR35","doi-asserted-by":"crossref","unstructured":"Li, X., Chen, J., Zhao, G., Pietikainen, M.: Remote heart rate measurement from face videos under realistic situations. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4264\u20134271 (2014)","DOI":"10.1109\/CVPR.2014.543"},{"key":"1720_CR36","doi-asserted-by":"crossref","unstructured":"Asthana, A., Zafeiriou, S., Cheng, S., Pantic, M.: Robust discriminative response map fitting with constrained local models. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3444\u20133451 (2013)","DOI":"10.1109\/CVPR.2013.442"},{"issue":"9","key":"1720_CR37","doi-asserted-by":"publisher","first-page":"1913","DOI":"10.1088\/0967-3334\/35\/9\/1913","volume":"35","author":"G De Haan","year":"2014","unstructured":"De Haan, G., Van Leest, A.: Improved motion robustness of remote-PPG by using the blood volume pulse signature. Physiol. Meas. 35(9), 1913\u20131913 (2014)","journal-title":"Physiol. Meas."},{"key":"1720_CR38","doi-asserted-by":"publisher","first-page":"1479","DOI":"10.1109\/TBME.2016.2609282","volume":"64","author":"W Wang","year":"2016","unstructured":"Wang, W., Den Brinker, A.C., Stuijk, S., De Haan, G.: Algorithmic principles of remote PPG. IEEE Trans. Biomed. Eng. 64, 1479\u20131491 (2016)","journal-title":"IEEE Trans. Biomed. Eng."},{"issue":"11","key":"1720_CR39","doi-asserted-by":"publisher","first-page":"5530","DOI":"10.1109\/JBHI.2023.3307942","volume":"27","author":"CA Casado","year":"2023","unstructured":"Casado, C.A., L\u00f3pez, M.B.: Face2PPG: an unsupervised pipeline for blood volume pulse extraction from faces. IEEE J. Biomed. Health Inform. 27(11), 5530\u20135541 (2023)","journal-title":"IEEE J. Biomed. Health Inform."},{"key":"1720_CR40","doi-asserted-by":"crossref","unstructured":"Pilz, C.S., Zaunseder, S., Krajewski, J., Blazek, V.: Local group invariance for heart rate estimation from face videos in the wild. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition Workshops, vol. 88, pp. 1254\u20131262 (2018)","DOI":"10.1109\/CVPRW.2018.00172"},{"key":"1720_CR41","doi-asserted-by":"crossref","unstructured":"Zhou, J., Guo, D., Wang, M.: Contrastive positive sample propagation along the audio-visual event line. IEEE Trans. Pattern Anal. Mach. Intell. (TPAMI) 7239\u20137257 (2023)","DOI":"10.1109\/TPAMI.2022.3223688"},{"key":"1720_CR42","doi-asserted-by":"crossref","unstructured":"Zhou, J., Guo, D., Zhong, Y., Wang, M.: Advancing weakly-supervised audio-visual video parsing via segment-wise pseudo labeling. Int. J. Comput. Vis. (IJCV) 1\u201322 (2024)","DOI":"10.1007\/s11263-024-02142-3"},{"key":"1720_CR43","doi-asserted-by":"crossref","unstructured":"Zhou, J., Shen, X., Wang, J., Zhang, J., Sun, W., Zhang, J., Birchfield, S., Guo, D., Kong, L., Wang, M., Zhong, Y.: Audio-visual segmentation with semantics. Int. J. Comput. Vis. 1\u201321 (2024)","DOI":"10.1007\/s11263-024-02261-x"},{"key":"1720_CR44","doi-asserted-by":"crossref","unstructured":"Zhou, J., Wang, J., Zhang, J., Sun, W., Zhang, J., Birchfield, S., Guo, D., Kong, L., Wang, M., Zhong, Y.: Audio\u2013visual segmentation. In: European Conference on Computer Vision (ECCV), pp. 386\u2013403 (2022)","DOI":"10.1007\/978-3-031-19836-6_22"},{"key":"1720_CR45","doi-asserted-by":"crossref","unstructured":"Zhou, J., Zheng, L., Zhong, Y., Hao, S., Wang, M.: Positive sample propagation along the audio-visual event line. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 8436\u20138444 (2021)","DOI":"10.1109\/CVPR46437.2021.00833"},{"key":"1720_CR46","doi-asserted-by":"crossref","unstructured":"Zhou, J., Guo, D., Mao, Y., Zhong, Y., Chang, X., Wang, M.: Label-anticipated event disentanglement for audio-visual video parsing. In: European Conference on Computer Vision (ECCV), pp. 1\u201322 (2024)","DOI":"10.1007\/978-3-031-72684-2_3"},{"key":"1720_CR47","doi-asserted-by":"crossref","unstructured":"Zhou, J., Guo, D., Guo, R., Mao, Y., Hu, J., Zhong, Y., Chang, X., Wang, M.: Towards open-vocabulary audio-visual event localization. arXiv preprint arXiv:2411.11278 (2024)","DOI":"10.1109\/CVPR52734.2025.00783"},{"key":"1720_CR48","doi-asserted-by":"crossref","unstructured":"Li, Z., Guo, D., Zhou, J., Zhang, J., Wang, M.: Object-aware adaptive-positivity learning for audio-visual question answering. In: Proceedings of the AAAI Conference on Artificial Intelligence (AAAI), pp. 3306\u20133314 (2024)","DOI":"10.1609\/aaai.v38i4.28116"},{"key":"1720_CR49","doi-asserted-by":"crossref","unstructured":"Li, Z., Zhou, J., Zhang, J., Tang, S., Li, K., Guo, D.: Patch-level sounding object tracking for audio-visual question answering. arXiv preprint arXiv:2412.10749 (2024)","DOI":"10.1609\/aaai.v39i5.32538"},{"key":"1720_CR50","doi-asserted-by":"crossref","unstructured":"Zhou, Z., Zhou, J., Qian, W., Tang, S., Chang, X., Guo, D.: Dense audio-visual event localization under cross-modal consistency and multi-temporal granularity collaboration. In: Proceedings of the AAAI Conference on Artificial Intelligence (AAAI), vol. 39(10), pp. 10905\u201310913 (2025)","DOI":"10.1609\/aaai.v39i10.33185"},{"key":"1720_CR51","doi-asserted-by":"crossref","unstructured":"Zhao, P., Zhou, J., Guo, D., Zhao, Y., Chen, Y.: Multimodal class-aware semantic enhancement network for audio-visual video parsing. arXiv preprint arXiv:2412.11248 (2024)","DOI":"10.1609\/aaai.v39i10.33134"},{"key":"1720_CR52","doi-asserted-by":"crossref","unstructured":"Shen, X., Li, D., Zhou, J., Qin, Z., He, B., Han, X., Li, A., Dai, Y., Kong, L., Wang, M., et al.: Fine-grained audible video description. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10585\u201310596 (2023)","DOI":"10.1109\/CVPR52729.2023.01020"},{"key":"1720_CR53","unstructured":"Guo, R., Ying, X., Chen, Y., Niu, D., Li, G., Qu, L., Qi, Y., Zhou, J., Xing, B., Yue, W., et al.: Audio-visual instance segmentation. arXiv preprint arXiv:2310.18709 (2023)"},{"key":"1720_CR54","doi-asserted-by":"crossref","unstructured":"Mao, Y., Shen, X., Zhang, J., Qin, Z., Zhou, J., Xiang, M., Zhong, Y., Dai, Y.: Tavgbench: benchmarking text to audible-video generation. In: Proceedings of the 32nd ACM International Conference on Multimedia, pp. 6607\u20136616 (2024)","DOI":"10.1145\/3664647.3680612"},{"key":"1720_CR55","unstructured":"Zhou, J., Guo, D., Zhong, Y., Wang, M.: Improving audio-visual video parsing with pseudo visual labels. arXiv preprint arXiv:2303.02344 (2023)"},{"issue":"7","key":"1720_CR56","doi-asserted-by":"publisher","first-page":"4388","DOI":"10.1109\/TCYB.2022.3175012","volume":"53","author":"P Song","year":"2022","unstructured":"Song, P., Guo, D., Zhou, J., Xu, M., Wang, M.: Memorial GAN with joint semantic optimization for unpaired image captioning. IEEE Trans. Cybern. 53(7), 4388\u20134399 (2022)","journal-title":"IEEE Trans. Cybern."},{"key":"1720_CR57","doi-asserted-by":"crossref","unstructured":"Narayanswamy, G., Liu, Y., Yang, Y., Ma, C., Liu, X., McDuff, D., Patel, S.: Bigsmall: efficient multi-task learning for disparate spatial and temporal physiological measurements. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 7914\u20137924 (2024)","DOI":"10.1109\/WACV57701.2024.00773"},{"key":"1720_CR58","unstructured":"Yu, Z., Li, X., Zhao, G.: Remote photoplethysmograph signal measurement from facial videos using spatio-temporal networks. In: Proceedings of the British Machine Vision Conference, pp. 1\u201312 (2019)"},{"key":"1720_CR59","doi-asserted-by":"crossref","unstructured":"Lu, H., Han, H., Zhou, S.K.: Dual-GAN: joint BVP and noise modeling for remote physiological measurement. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12404\u201312413 (2021)","DOI":"10.1109\/CVPR46437.2021.01222"},{"key":"1720_CR60","doi-asserted-by":"crossref","unstructured":"Qian, W., Guo, D., Li, K., Zhang, X., Tian, X., Yang, X., Wang, M.: Dual-path tokenlearner for remote photoplethysmography-based physiological measurement with facial videos. IEEE Trans. Comput. Soc. Syst. (2024)","DOI":"10.1109\/TCSS.2024.3356713"},{"key":"1720_CR61","doi-asserted-by":"crossref","unstructured":"Lu, H., Yu, Z., Niu, X., Chen, Y.-C.: Neuron structure modeling for generalizable remote physiological measurement. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18589\u201318599 (2023)","DOI":"10.1109\/CVPR52729.2023.01783"},{"key":"1720_CR62","unstructured":"Qian, W., Li, Q., Li, K., Wang, X., Sun, X., Wang, M., Guo, D.: Joint spatial-temporal modeling and contrastive learning for self-supervised heart rate measurement. arXiv preprint arXiv:2406.04942 (2024)"},{"key":"1720_CR63","doi-asserted-by":"crossref","unstructured":"Gideon, J., Stent, S.: The way to my heart is through contrastive learning: remote photoplethysmography from unlabelled video. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3995\u20134004 (2021)","DOI":"10.1109\/ICCV48922.2021.00396"},{"key":"1720_CR64","doi-asserted-by":"crossref","unstructured":"Sun, Z., Li, X.: Contrast-phys+: unsupervised and weakly-supervised video-based remote physiological measurement via spatiotemporal contrast. IEEE Trans. Pattern Anal. Mach. Intell. (2024)","DOI":"10.1109\/TPAMI.2024.3367910"},{"key":"1720_CR65","doi-asserted-by":"publisher","first-page":"7278","DOI":"10.1109\/TMM.2024.3363660","volume":"26","author":"X Liu","year":"2024","unstructured":"Liu, X., Zhang, Y., Yu, Z., Lu, H., Yue, H., Yang, J.: RPPG-MAE: self-supervised pretraining with masked autoencoders for remote physiological measurements. IEEE Trans. Multimedia 26, 7278\u20137293 (2024)","journal-title":"IEEE Trans. Multimedia"},{"key":"1720_CR66","doi-asserted-by":"crossref","unstructured":"Speth, J., Vance, N., Flynn, P., Czajka, A.: Non-contrastive unsupervised learning of physiological signals from video. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14464\u201314474 (2023)","DOI":"10.1109\/CVPR52729.2023.01390"},{"key":"1720_CR67","doi-asserted-by":"crossref","unstructured":"Li, Z., Yin, L.: Contactless pulse estimation leveraging pseudo labels and self-supervision. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 20588\u201320597 (2023)","DOI":"10.1109\/ICCV51070.2023.01882"},{"issue":"11","key":"1720_CR68","first-page":"13844","volume":"45","author":"Z Yue","year":"2023","unstructured":"Yue, Z., Shi, M., Ding, S.: Facial video-based remote physiological measurement via self-supervised learning. IEEE Trans. Pattern Anal. Mach. Intell. 45(11), 13844\u201313859 (2023)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1720_CR69","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. Adv. Neural. Inf. Process. Syst. 33, 6840\u20136851 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1720_CR70","unstructured":"Song, J., Meng, C., Ermon, S.: Denoising diffusion implicit models. arXiv preprint arXiv:2010.02502 (2020)"},{"key":"1720_CR71","doi-asserted-by":"crossref","unstructured":"Avrahami, O., Lischinski, D., Fried, O.: Blended diffusion for text-driven editing of natural images. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18208\u201318218 (2022)","DOI":"10.1109\/CVPR52688.2022.01767"},{"key":"1720_CR72","doi-asserted-by":"crossref","unstructured":"Fan, W.-C., Chen, Y.-C., Chen, D., Cheng, Y., Yuan, L., Wang, Y.-C.F.: Frido: feature pyramid diffusion for complex scene image synthesis. arXiv preprint arXiv:2208.13753 (2022)","DOI":"10.1609\/aaai.v37i1.25133"},{"key":"1720_CR73","doi-asserted-by":"crossref","unstructured":"Huang, R., Zhao, Z., Liu, H., Liu, J., Cui, C., Ren, Y.: Prodiff: progressive fast diffusion model for high-quality text-to-speech. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 2595\u20132605 (2022)","DOI":"10.1145\/3503161.3547855"},{"key":"1720_CR74","unstructured":"Kim, S., Kim, H., Yoon, S.: Guided-tts 2: a diffusion model for high-quality adaptive text-to-speech with untranscribed data. arXiv preprint arXiv:2205.15370 (2022)"},{"key":"1720_CR75","unstructured":"Vignac, C., Krawczuk, I., Siraudin, A., Wang, B., Cevher, V., Frossard, P.: Digress: discrete denoising diffusion for graph generation. arXiv preprint arXiv:2209.14734 (2022)"},{"key":"1720_CR76","unstructured":"Niu, C., Song, Y., Song, J., Zhao, S., Grover, A., Ermon, S.: Permutation invariant graph generation via score-based generative modeling. In: International Conference on Artificial Intelligence and Statistics, pp. 4474\u20134484. PMLR (2020)"},{"key":"1720_CR77","unstructured":"Jo, J., Lee, S., Hwang, S.J.: Score-based generative modeling of graphs via the system of stochastic differential equations. In: International Conference on Machine Learning, pp. 10362\u201310383. PMLR (2022)"},{"key":"1720_CR78","unstructured":"Yan, Q., Liang, Z., Song, Y., Liao, R., Wang, L.: Swingnn: rethinking permutation invariance in diffusion models for graph generation. arXiv preprint arXiv:2307.01646 (2023)"},{"key":"1720_CR79","doi-asserted-by":"crossref","unstructured":"Tulyakov, S., Liu, M.-Y., Yang, X., Kautz, J.: Mocogan: decomposing motion and content for video generation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1526\u20131535 (2018)","DOI":"10.1109\/CVPR.2018.00165"},{"key":"1720_CR80","unstructured":"Ho, J., Chan, W., Saharia, C., Whang, J., Gao, R., Gritsenko, A., Kingma, D.P., Poole, B., Norouzi, M., Fleet, D.J., et al.: Imagen video: high definition video generation with diffusion models. arXiv preprint arXiv:2210.02303 (2022)"},{"key":"1720_CR81","unstructured":"Singer, U., Polyak, A., Hayes, T., Yin, X., An, J., Zhang, S., Hu, Q., Yang, H., Ashual, O., Gafni, O., et al.: Make-a-video: text-to-video generation without text-video data. arXiv preprint arXiv:2209.14792 (2022)"},{"key":"1720_CR82","doi-asserted-by":"crossref","unstructured":"Brempong, E.A., Kornblith, S., Chen, T., Parmar, N., Minderer, M., Norouzi, M.: Denoising pretraining for semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4175\u20134186 (2022)","DOI":"10.1109\/CVPRW56347.2022.00462"},{"key":"1720_CR83","unstructured":"Baranchuk, D., Rubachev, I., Voynov, A., Khrulkov, V., Babenko, A.: Label-efficient semantic segmentation with diffusion models. arXiv preprint arXiv:2112.03126 (2021)"},{"key":"1720_CR84","doi-asserted-by":"crossref","unstructured":"Chen, S., Sun, P., Song, Y., Luo, P.: Diffusiondet: diffusion model for object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 19830\u201319843 (2023)","DOI":"10.1109\/ICCV51070.2023.01816"},{"key":"1720_CR85","doi-asserted-by":"crossref","unstructured":"Fang, H., Han, B., Zhang, S., Zhou, S., Hu, C., Ye, W.-M.: Data augmentation for object detection via controllable diffusion models. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 1257\u20131266 (2024)","DOI":"10.1109\/WACV57701.2024.00129"},{"key":"1720_CR86","doi-asserted-by":"publisher","first-page":"2903","DOI":"10.1109\/TIP.2019.2954209","volume":"29","author":"P Jiang","year":"2019","unstructured":"Jiang, P., Pan, Z., Tu, C., Vasconcelos, N., Chen, B., Peng, J.: Super diffusion for salient object detection. IEEE Trans. Image Process. 29, 2903\u20132917 (2019)","journal-title":"IEEE Trans. Image Process."},{"key":"1720_CR87","doi-asserted-by":"crossref","unstructured":"Ceylan, D., Huang, C.-H.P., Mitra, N.J.: Pix2video: video editing using image diffusion. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 23206\u201323217 (2023)","DOI":"10.1109\/ICCV51070.2023.02121"},{"key":"1720_CR88","doi-asserted-by":"crossref","unstructured":"Chai, W., Guo, X., Wang, G., Lu, Y.: Stablevideo: text-driven consistency-aware diffusion video editing. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 23040\u201323050 (2023)","DOI":"10.1109\/ICCV51070.2023.02106"},{"key":"1720_CR89","doi-asserted-by":"crossref","unstructured":"Shan, W., Liu, Z., Zhang, X., Wang, Z., Han, K., Wang, S., Ma, S., Gao, W.: Diffusion-based 3d human pose estimation with multi-hypothesis aggregation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 14761\u201314771 (2023)","DOI":"10.1109\/ICCV51070.2023.01356"},{"issue":"8","key":"1720_CR90","doi-asserted-by":"publisher","first-page":"743","DOI":"10.3390\/bioengineering11080743","volume":"11","author":"S Chen","year":"2024","unstructured":"Chen, S., Wong, K.-L., Chin, J.-W., Chan, T.-T., So, R.H.: Diffphys: enhancing signal-to-noise ratio in remote photoplethysmography signal using a diffusion model approach. Bioengineering 11(8), 743 (2024)","journal-title":"Bioengineering"},{"issue":"26","key":"1720_CR91","doi-asserted-by":"publisher","first-page":"21434","DOI":"10.1364\/OE.16.021434","volume":"16","author":"W Verkruysse","year":"2008","unstructured":"Verkruysse, W.: Remote plethysmographic imaging using ambient light. Opt. Exp. 16(26), 21434\u201321445 (2008)","journal-title":"Opt. Exp."},{"issue":"10","key":"1720_CR92","doi-asserted-by":"publisher","first-page":"10762","DOI":"10.1364\/OE.18.010762","volume":"18","author":"M-Z Poh","year":"2010","unstructured":"Poh, M.-Z., McDuff, D.J., Picard, R.W.: Non-contact, automated cardiac pulse measurements using video imaging and blind source separation. Opt. Exp. 18(10), 10762\u201310774 (2010)","journal-title":"Opt. Exp."},{"key":"1720_CR93","doi-asserted-by":"crossref","unstructured":"Qian, W., Li, K., Guo, D., Hu, B., Wang, M.: Cluster-phys: facial clues clustering towards efficient remote physiological measurement. In: Proceedings of the 32nd ACM International Conference on Multimedia, pp. 330\u2013339 (2024)","DOI":"10.1145\/3664647.3680670"},{"key":"1720_CR94","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1016\/j.patrec.2017.10.017","volume":"124","author":"S Bobbia","year":"2019","unstructured":"Bobbia, S., Macwan, R., Benezeth, Y., Mansouri, A., Dubois, J.: Unsupervised skin tissue segmentation for remote photoplethysmography. Pattern Recogn. Lett. 124, 82\u201390 (2019)","journal-title":"Pattern Recogn. Lett."},{"key":"1720_CR95","doi-asserted-by":"crossref","unstructured":"Stricker, R., M\u00fcller, S., Gross, H.-M.: Non-contact video-based pulse rate measurement on a mobile service robot. In: Proc. IISRHIC, pp. 1056\u20131062 (2014)","DOI":"10.1109\/ROMAN.2014.6926392"},{"key":"1720_CR96","doi-asserted-by":"crossref","unstructured":"Tulyakov, S., Alameda-Pineda, X., Ricci, E., Yin, L., Cohn, J.F., Sebe, N.: Self-adaptive matrix completion for heart rate estimation from face videos under realistic conditions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2396\u20132404 (2016)","DOI":"10.1109\/CVPR.2016.263"},{"key":"1720_CR97","doi-asserted-by":"crossref","unstructured":"Tang, J., Chen, K., Wang, Y., Shi, Y., Patel, S., McDuff, D., Liu, X.: MMPD: multi-domain mobile video physiology dataset. In: 2023 45th Annual International Conference of the IEEE Engineering in Medicine & Biology Society (EMBC), pp. 1\u20135. IEEE (2023)","DOI":"10.1109\/EMBC40787.2023.10340857"},{"key":"1720_CR98","doi-asserted-by":"crossref","unstructured":"Tsou, Y.-Y., Lee, Y.-A., Hsu, C.-T., Chang, S.-H.: Siamese-rppg network: Remote photoplethysmography signal estimation from face videos. In: Proceedings of the 35th Annual ACM Symposium on Applied Computing, pp. 2066\u20132073 (2020)","DOI":"10.1145\/3341105.3373905"},{"issue":"5","key":"1720_CR99","doi-asserted-by":"publisher","first-page":"1373","DOI":"10.1109\/JBHI.2021.3051176","volume":"25","author":"R Song","year":"2021","unstructured":"Song, R., Chen, H., Cheng, J., Li, C., Liu, Y., Chen, X.: Pulsegan: learning to generate realistic pulse waveforms in remote photoplethysmography. IEEE J. Biomed. Health Inform. 25(5), 1373\u20131384 (2021)","journal-title":"IEEE J. Biomed. Health Inform."},{"key":"1720_CR100","doi-asserted-by":"crossref","unstructured":"Sun, Z., Li, X.: Contrast-phys: unsupervised video-based remote physiological measurement via spatiotemporal contrast. In: Proceedings of the European Conference on Computer Vision, pp. 492\u2013510 (2022)","DOI":"10.1007\/978-3-031-19775-8_29"},{"key":"1720_CR101","doi-asserted-by":"crossref","unstructured":"Li, J., Yu, Z., Shi, J.: Learning motion-robust remote photoplethysmography through arbitrary resolution videos. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp. 1\u201311 (2023)","DOI":"10.1609\/aaai.v37i1.25217"}],"container-title":["Machine Vision and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00138-025-01720-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00138-025-01720-w\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00138-025-01720-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,12]],"date-time":"2025-09-12T14:58:41Z","timestamp":1757689121000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00138-025-01720-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,29]]},"references-count":101,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2025,9]]}},"alternative-id":["1720"],"URL":"https:\/\/doi.org\/10.1007\/s00138-025-01720-w","relation":{},"ISSN":["0932-8092","1432-1769"],"issn-type":[{"value":"0932-8092","type":"print"},{"value":"1432-1769","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,7,29]]},"assertion":[{"value":"27 February 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 May 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 June 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 July 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"105"}}