{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,11]],"date-time":"2025-09-11T19:13:24Z","timestamp":1757618004410,"version":"3.44.0"},"reference-count":62,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2025,5,9]],"date-time":"2025-05-09T00:00:00Z","timestamp":1746748800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,5,9]],"date-time":"2025-05-09T00:00:00Z","timestamp":1746748800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Machine Vision and Applications"],"published-print":{"date-parts":[[2025,7]]},"DOI":"10.1007\/s00138-025-01699-4","type":"journal-article","created":{"date-parts":[[2025,5,9]],"date-time":"2025-05-09T13:59:40Z","timestamp":1746799180000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Speech-aided facial video super resolution with accurate lip motion and enhanced frequency details"],"prefix":"10.1007","volume":"36","author":[{"given":"Shailza","family":"Sharma","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Vivek","family":"Singh","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Abhinav","family":"Dhall","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Vinay","family":"Kumar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,9]]},"reference":[{"key":"1699_CR1","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2021.115780","volume":"186","author":"S Sharma","year":"2021","unstructured":"Sharma, S., Kumar, V.: An efficient image super resolution model with dense skip connections between complex filter structures in generative adversarial networks. Expert Syst. Appl. 186, 115780 (2021)","journal-title":"Expert Syst. Appl."},{"key":"1699_CR2","doi-asserted-by":"crossref","unstructured":"Sharma, S., Dhall, A., Kumar, D.V., Singh, V.: Dual stage semantic information based generative adversarial network for image super-resolution*. In: Proceedings of the Fourteenth Indian Conference on Computer Vision, Graphics and Image Processing, pp. 1\u20139 (2023)","DOI":"10.1145\/3627631.3627646"},{"issue":"6","key":"1699_CR3","doi-asserted-by":"publisher","first-page":"581","DOI":"10.1049\/bme2.12029","volume":"10","author":"D Zeng","year":"2021","unstructured":"Zeng, D., Veldhuis, R., Spreeuwers, L.: A survey of face recognition techniques under occlusion. IET Biom. 10(6), 581\u2013606 (2021)","journal-title":"IET Biom."},{"key":"1699_CR4","doi-asserted-by":"crossref","unstructured":"Yadav, D., Salmani, S.: Deepfake: a survey on facial forgery technique using generative adversarial network. In: 2019 International Conference on Intelligent Computing and Control Systems (ICCS), pp. 852\u2013857. IEEE (2019)","DOI":"10.1109\/ICCS45141.2019.9065881"},{"key":"1699_CR5","doi-asserted-by":"crossref","unstructured":"Remya\u00a0Revi, K., Vidya, K., Wilscy, M.: Detection of deepfake images created using generative adversarial networks: A review. In: Second International Conference on Networks and Advances in Computational Technologies, pp. 25\u201335. Springer (2021)","DOI":"10.1007\/978-3-030-49500-8_3"},{"key":"1699_CR6","doi-asserted-by":"crossref","unstructured":"Cai, J., Meng, Z., Khan, A.S., O\u2019Reilly, J., Li, Z., Han, S., Tong, Y.: Identity-free facial expression recognition using conditional generative adversarial network. In: 2021 IEEE International Conference on Image Processing (ICIP), pp. 1344\u20131348. IEEE (2021)","DOI":"10.1109\/ICIP42928.2021.9506593"},{"issue":"12","key":"1699_CR7","doi-asserted-by":"publisher","first-page":"9061","DOI":"10.1007\/s00521-018-3867-5","volume":"31","author":"VS Bawa","year":"2019","unstructured":"Bawa, V.S., Kumar, V.: Emotional sentiment analysis for a group of people based on transfer learning with a multi-modal system. Neural Comput. Appl. 31(12), 9061\u20139072 (2019)","journal-title":"Neural Comput. Appl."},{"key":"1699_CR8","doi-asserted-by":"publisher","first-page":"132","DOI":"10.1016\/j.neunet.2020.09.001","volume":"133","author":"J Lin","year":"2021","unstructured":"Lin, J., Li, Y., Yang, G.: Fpgan: face de-identification method with generative adversarial networks for social robots. Neural Netw. 133, 132\u2013147 (2021)","journal-title":"Neural Netw."},{"key":"1699_CR9","doi-asserted-by":"publisher","first-page":"261","DOI":"10.1016\/j.neunet.2019.12.009","volume":"123","author":"J Wan","year":"2020","unstructured":"Wan, J., Li, J., Lai, Z., Du, B., Zhang, L.: Robust face alignment by cascaded regression and de-occlusion. Neural Netw. 123, 261\u2013272 (2020)","journal-title":"Neural Netw."},{"key":"1699_CR10","doi-asserted-by":"publisher","first-page":"110421","DOI":"10.1109\/ACCESS.2021.3102042","volume":"9","author":"VS Bawa","year":"2021","unstructured":"Bawa, V.S., Sharma, S., Usman, M., Gupta, A., Kumar, V.: An automatic multimedia likability prediction system based on facial expression of observer. IEEE Access 9, 110421\u2013110434 (2021)","journal-title":"IEEE Access"},{"key":"1699_CR11","doi-asserted-by":"crossref","unstructured":"Hu, X., Ren, W., LaMaster, J., Cao, X., Li, X., Li, Z., Menze, B., Liu, W.: Face super-resolution guided by 3d facial priors. In: European Conference on Computer Vision, 763\u2013780. Springer (2020)","DOI":"10.1007\/978-3-030-58548-8_44"},{"key":"1699_CR12","doi-asserted-by":"crossref","unstructured":"He, J., Shi, W., Chen, K., Fu, L., Dong, C.: Gcfsr: a generative and controllable face super resolution method without facial and gan priors. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1889\u20131898 (2022)","DOI":"10.1109\/CVPR52688.2022.00193"},{"issue":"1","key":"1699_CR13","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1186\/s13638-020-01760-y","volume":"2020","author":"Z Fan","year":"2020","unstructured":"Fan, Z., Hu, X., Chen, C., Wang, X., Peng, S.: Facial image super-resolution guided by adaptive geometric features. EURASIP J. Wirel. Commun. Netw. 2020(1), 1\u201315 (2020)","journal-title":"EURASIP J. Wirel. Commun. Netw."},{"key":"1699_CR14","doi-asserted-by":"crossref","unstructured":"Qiu, Z., Yao, T., Mei, T.: Learning spatio-temporal representation with pseudo-3d residual networks. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 5533\u20135541 (2017)","DOI":"10.1109\/ICCV.2017.590"},{"key":"1699_CR15","unstructured":"Kim, S.Y., Lim, J., Na, T., Kim, M.: 3dsrnet: video super-resolution using 3d convolutional neural networks. arXiv preprint arXiv:1812.09079 (2018)"},{"key":"1699_CR16","doi-asserted-by":"crossref","unstructured":"Caballero, J., Ledig, C., Aitken, A., Acosta, A., Totz, J., Wang, Z., Shi, W.: Real-time video super-resolution with spatio-temporal networks and motion compensation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4778\u20134787 (2017)","DOI":"10.1109\/CVPR.2017.304"},{"key":"1699_CR17","doi-asserted-by":"crossref","unstructured":"Jo, Y., Oh, S.W., Kang, J., Kim, S.J.: Deep video super-resolution network using dynamic upsampling filters without explicit motion compensation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3224\u20133232 (2018)","DOI":"10.1109\/CVPR.2018.00340"},{"key":"1699_CR18","unstructured":"Xu, Z.-Q.J., Zhang, Y., Luo, T., Xiao, Y., Ma, Z.: Frequency principle: Fourier analysis sheds light on deep neural networks. arXiv preprint arXiv:1901.06523 (2019)"},{"key":"1699_CR19","unstructured":"Afouras, T., Chung, J.S., Senior, A., Vinyals, O., Zisserman, A.: Deep audio-visual speech recognition. IEEE transactions on pattern analysis and machine intelligence (2018)"},{"key":"1699_CR20","doi-asserted-by":"crossref","unstructured":"Oh, T.-H., Dekel, T., Kim, C., Mosseri, I., Freeman, W.T., Rubinstein, M., Matusik, W.: Speech2face: Learning the face behind a voice. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7539\u20137548 (2019)","DOI":"10.1109\/CVPR.2019.00772"},{"key":"1699_CR21","doi-asserted-by":"crossref","unstructured":"Wang, H., Zhu, Y., Green, B., Adam, H., Yuille, A., Chen, L.-C.: Axial-deeplab: stand-alone axial-attention for panoptic segmentation. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part IV, pp. 108\u2013126. Springer (2020)","DOI":"10.1007\/978-3-030-58548-8_7"},{"key":"1699_CR22","doi-asserted-by":"crossref","unstructured":"Yu, X., Porikli, F.: Ultra-resolving face images by discriminative generative networks. In: European Conference on Computer Vision, pp. 318\u2013333. Springer (2016)","DOI":"10.1007\/978-3-319-46454-1_20"},{"key":"1699_CR23","unstructured":"Chen, Z., Tong, Y.: Face super-resolution through wasserstein gans. arXiv preprint arXiv:1705.02438 (2017)"},{"key":"1699_CR24","doi-asserted-by":"crossref","unstructured":"Ko, S., Dai, B.-R.: Multi-Laplacian GAN with edge enhancement for face super resolution. In: 2020 25th International Conference on Pattern Recognition (ICPR), pp. 3505\u20133512. IEEE (2021)","DOI":"10.1109\/ICPR48806.2021.9412950"},{"issue":"2","key":"1699_CR25","doi-asserted-by":"publisher","first-page":"442","DOI":"10.1049\/ipr2.12359","volume":"16","author":"A Aakerberg","year":"2022","unstructured":"Aakerberg, A., Nasrollahi, K., Moeslund, T.B.: Real-world super-resolution of face-images from surveillance cameras. IET Image Proc. 16(2), 442\u2013452 (2022)","journal-title":"IET Image Proc."},{"key":"1699_CR26","doi-asserted-by":"crossref","unstructured":"Chan, K.C.K., Zhou, S., Xu, X., Loy, C.C.: Investigating tradeoffs in real-world video super-resolution. In: IEEE Conference on Computer Vision and Pattern Recognition (2022)","DOI":"10.1109\/CVPR52688.2022.00587"},{"key":"1699_CR27","doi-asserted-by":"crossref","unstructured":"Ledig, C., Theis, L., Husz\u00e1r, F., Caballero, J., Cunningham, A., Acosta, A., Aitken, A., Tejani, A., Totz, J., Wang, Z., et al.: Photo-realistic single image super-resolution using a generative adversarial network. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4681\u20134690 (2017)","DOI":"10.1109\/CVPR.2017.19"},{"key":"1699_CR28","doi-asserted-by":"crossref","unstructured":"Wang, Y., Isobe, T., Jia, X., Tao, X., Lu, H., Tai, Y.-W.: Compression-aware video super-resolution. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2012\u20132021 (2023)","DOI":"10.1109\/CVPR52729.2023.00200"},{"key":"1699_CR29","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2023.110059","volume":"146","author":"R Chen","year":"2024","unstructured":"Chen, R., Mu, Y., Zhang, Y.: High-order relational generative adversarial network for video super-resolution. Pattern Recogn. 146, 110059 (2024)","journal-title":"Pattern Recogn."},{"issue":"3","key":"1699_CR30","doi-asserted-by":"publisher","first-page":"129","DOI":"10.1016\/j.tics.2004.01.008","volume":"8","author":"P Belin","year":"2004","unstructured":"Belin, P., Fecteau, S., Bedard, C.: Thinking the voice: neural correlates of voice perception. Trends Cogn. Sci. 8(3), 129\u2013135 (2004)","journal-title":"Trends Cogn. Sci."},{"key":"1699_CR31","doi-asserted-by":"crossref","unstructured":"Hu, D., Nie, F., Li, X.: Deep multimodal clustering for unsupervised audiovisual learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9248\u20139257 (2019)","DOI":"10.1109\/CVPR.2019.00947"},{"key":"1699_CR32","doi-asserted-by":"crossref","unstructured":"Tian, Y., Li, D., Xu, C.: Unified multisensory perception: weakly-supervised audio-visual video parsing. In: European Conference on Computer Vision, pp. 436\u2013454. Springer (2020)","DOI":"10.1007\/978-3-030-58580-8_26"},{"key":"1699_CR33","doi-asserted-by":"crossref","unstructured":"Xuan, H., Zhang, Z., Chen, S., Yang, J., Yan, Y.: Cross-modal attention network for temporal inconsistent audio-visual event localization. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, 279\u2013286 (2020)","DOI":"10.1609\/aaai.v34i01.5361"},{"key":"1699_CR34","doi-asserted-by":"crossref","unstructured":"Wu, Y., Zhu, L., Yan, Y., Yang, Y.: Dual attention matching for audio-visual event localization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6292\u20136300 (2019)","DOI":"10.1109\/ICCV.2019.00639"},{"key":"1699_CR35","unstructured":"Wen, Y., Raj, B., Singh, R.: Face reconstruction from voice using generative adversarial networks. In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"key":"1699_CR36","doi-asserted-by":"crossref","unstructured":"Chen, L., Maddox, R.K., Duan, Z., Xu, C.: Hierarchical cross-modal talking face generation with dynamic pixel-wise loss. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7832\u20137841 (2019)","DOI":"10.1109\/CVPR.2019.00802"},{"key":"1699_CR37","doi-asserted-by":"crossref","unstructured":"Zhang, X., Wu, X., Zhai, X., Ben, X., Tu, C.: Davd-net: deep audio-aided video decompression of talking heads. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12335\u201312344 (2020)","DOI":"10.1109\/CVPR42600.2020.01235"},{"key":"1699_CR38","doi-asserted-by":"crossref","unstructured":"Meishvili, G., Jenni, S., Favaro, P.: Learning to have an ear for face super-resolution. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1364\u20131374 (2020)","DOI":"10.1109\/CVPR42600.2020.00144"},{"key":"1699_CR39","doi-asserted-by":"crossref","unstructured":"Hong, F.-T., Shen, L., Xu, D.: Dagan++: depth-aware generative adversarial network for talking head video generation. IEEE Trans. Pattern Anal. Mach. Intell 46(5), 2997\u20133012 (2023)","DOI":"10.1109\/TPAMI.2023.3339964"},{"key":"1699_CR40","doi-asserted-by":"crossref","unstructured":"Hwang, G., Hong, S., Lee, S., Park, S., Chae, G.: Discohead: audio-and-video-driven talking head generation by disentangled control of head pose and facial expressions. In: ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1\u20135. IEEE (2023)","DOI":"10.1109\/ICASSP49357.2023.10095670"},{"key":"1699_CR41","doi-asserted-by":"crossref","unstructured":"Zhang, B., Qi, C., Zhang, P., Zhang, B., Wu, H., Chen, D., Chen, Q., Wang, Y., Wen, F.: Metaportrait: Identity-preserving talking head generation with fast personalized adaptation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 22096\u201322105 (2023)","DOI":"10.1109\/CVPR52729.2023.02116"},{"key":"1699_CR42","doi-asserted-by":"crossref","unstructured":"Hong, F.-T., Xu, D.: Implicit identity representation conditioned memory compensation network for talking head video generation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 23062\u201323072 (2023)","DOI":"10.1109\/ICCV51070.2023.02108"},{"key":"1699_CR43","doi-asserted-by":"crossref","unstructured":"Shi, W., Caballero, J., Husz\u00e1r, F., Totz, J., Aitken, A.P., Bishop, R., Rueckert, D., Wang, Z.: Real-time single image and video super-resolution using an efficient sub-pixel convolutional neural network. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1874\u20131883 (2016)","DOI":"10.1109\/CVPR.2016.207"},{"key":"1699_CR44","doi-asserted-by":"crossref","unstructured":"Sharma, S., Dhall, A., Kumar, V.: Frequency aware face hallucination generative adversarial network with semantic structural constraint. Comput. Vis. Image Underst. 223,103553 (2022). https:\/\/www.sciencedirect.com\/science\/article\/pii\/S107731422200131X","DOI":"10.1016\/j.cviu.2022.103553"},{"key":"1699_CR45","doi-asserted-by":"crossref","unstructured":"Chollet, F.: Xception: Deep learning with depthwise separable convolutions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1251\u20131258 (2017)","DOI":"10.1109\/CVPR.2017.195"},{"key":"1699_CR46","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., Sun, G.: Squeeze-and-excitation networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7132\u20137141 (2018)","DOI":"10.1109\/CVPR.2018.00745"},{"key":"1699_CR47","doi-asserted-by":"crossref","unstructured":"Martinez, B., Ma, P., Petridis, S., Pantic, M.: Lipreading using temporal convolutional networks. In: ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6319\u20136323. IEEE (2020)","DOI":"10.1109\/ICASSP40776.2020.9053841"},{"issue":"10","key":"1699_CR48","doi-asserted-by":"publisher","first-page":"3","DOI":"10.23915\/distill.00003","volume":"1","author":"A Odena","year":"2016","unstructured":"Odena, A., Dumoulin, V., Olah, C.: Deconvolution and checkerboard artifacts. Distill 1(10), 3 (2016)","journal-title":"Distill"},{"key":"1699_CR49","doi-asserted-by":"crossref","unstructured":"Jiang, L., Dai, B., Wu, W., Loy, C.C.: Focal frequency loss for image reconstruction and synthesis. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13919\u201313929 (2021)","DOI":"10.1109\/ICCV48922.2021.01366"},{"issue":"1","key":"1699_CR50","doi-asserted-by":"publisher","first-page":"99","DOI":"10.1145\/3503250","volume":"65","author":"B Mildenhall","year":"2021","unstructured":"Mildenhall, B., Srinivasan, P.P., Tancik, M., Barron, J.T., Ramamoorthi, R., Ng, R.: Nerf: representing scenes as neural radiance fields for view synthesis. Commun. ACM 65(1), 99\u2013106 (2021)","journal-title":"Commun. ACM"},{"key":"1699_CR51","unstructured":"Rahaman, N., Baratin, A., Arpit, D., Draxler, F., Lin, M., Hamprecht, F., Bengio, Y., Courville, A.: On the spectral bias of neural networks. In: International Conference on Machine Learning, pp. 5301\u20135310. PMLR (2019)"},{"key":"1699_CR52","doi-asserted-by":"crossref","unstructured":"Yang, S., Zhang, Y., Feng, D., Yang, M., Wang, C., Xiao, J., Long, K., Shan, S., Chen, X.: Lrw-1000: A naturally-distributed large-scale benchmark for lip reading in the wild. In: 2019 14th IEEE International Conference on Automatic Face & Gesture Recognition (FG 2019), pp. 1\u20138. IEEE (2019)","DOI":"10.1109\/FG.2019.8756582"},{"key":"1699_CR53","unstructured":"Amos, B., Ludwiczuk, B., Satyanarayanan, M.: Openface: a general-purpose face recognition library with mobile applications. Technical report, CMU-CS-16-118, CMU School of Computer Science (2016)"},{"issue":"6","key":"1699_CR54","doi-asserted-by":"publisher","first-page":"523","DOI":"10.1121\/1.5042758","volume":"143","author":"N Alghamdi","year":"2018","unstructured":"Alghamdi, N., Maddock, S., Marxer, R., Barker, J., Brown, G.J.: A corpus of audio-visual Lombard speech with frontal and profile views. J. Acoust. Soc. Am. 143(6), 523\u2013529 (2018)","journal-title":"J. Acoust. Soc. Am."},{"key":"1699_CR55","doi-asserted-by":"crossref","unstructured":"Xie, L., Wang, X., Zhang, H., Dong, C., Shan, Y.: Vfhq: a high-quality dataset and benchmark for video face super-resolution. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 657\u2013666 (2022)","DOI":"10.1109\/CVPRW56347.2022.00081"},{"key":"1699_CR56","doi-asserted-by":"crossref","unstructured":"Nagrani, A., Chung, J.S., Zisserman, A.: Voxceleb: a large-scale speaker identification dataset. arXiv preprint arXiv:1706.08612 (2017)","DOI":"10.21437\/Interspeech.2017-950"},{"key":"1699_CR57","doi-asserted-by":"crossref","unstructured":"Lim, B., Son, S., Kim, H., Nah, S., Mu Lee, K.: Enhanced deep residual networks for single image super-resolution. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition Workshops, pp. 136\u2013144 (2017)","DOI":"10.1109\/CVPRW.2017.151"},{"key":"1699_CR58","unstructured":"Kim, D., Kim, M., Kwon, G., Kim, D.-S.: Progressive face super-resolution via attention to facial landmark. arXiv preprint arXiv:1908.08239 (2019)"},{"key":"1699_CR59","doi-asserted-by":"crossref","unstructured":"Zhang, K., Zhang, Z., Cheng, C.-W., Hsu, W.H., Qiao, Y., Liu, W., Zhang, T.: Super-identity convolutional neural network for face hallucination. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 183\u2013198 (2018)","DOI":"10.1007\/978-3-030-01252-6_12"},{"key":"1699_CR60","doi-asserted-by":"crossref","unstructured":"Wang, X., Li, Y., Zhang, H., Shan, Y.: Towards real-world blind face restoration with generative facial prior. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9168\u20139178 (2021)","DOI":"10.1109\/CVPR46437.2021.00905"},{"key":"1699_CR61","doi-asserted-by":"crossref","unstructured":"Chan, K.C.K., Zhou, S., Xu, X., Loy, C.C.: BasicVSR++: Improving video super-resolution with enhanced propagation and alignment. In: IEEE Conference on Computer Vision and Pattern Recognition (2022)","DOI":"10.1109\/CVPR52688.2022.00588"},{"key":"1699_CR62","doi-asserted-by":"crossref","unstructured":"Zhou, S., Yang, P., Wang, J., Luo, Y., Loy, C.C.: Upscale-a-video: Temporal-consistent diffusion model for real-world video super-resolution. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2535\u20132545 (2024)","DOI":"10.1109\/CVPR52733.2024.00245"}],"container-title":["Machine Vision and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00138-025-01699-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00138-025-01699-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00138-025-01699-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T14:36:56Z","timestamp":1757169416000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00138-025-01699-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,9]]},"references-count":62,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2025,7]]}},"alternative-id":["1699"],"URL":"https:\/\/doi.org\/10.1007\/s00138-025-01699-4","relation":{},"ISSN":["0932-8092","1432-1769"],"issn-type":[{"type":"print","value":"0932-8092"},{"type":"electronic","value":"1432-1769"}],"subject":[],"published":{"date-parts":[[2025,5,9]]},"assertion":[{"value":"12 June 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 January 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 April 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 May 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"We obtained informed consent for all datasets and images used in this article, adhering to ethical standards that ensure the privacy and integrity of the data.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical considerations and informed consent"}}],"article-number":"77"}}