{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,20]],"date-time":"2026-04-20T18:37:43Z","timestamp":1776710263760,"version":"3.51.2"},"reference-count":62,"publisher":"Springer Science and Business Media LLC","issue":"11","license":[{"start":{"date-parts":[[2025,5,23]],"date-time":"2025-05-23T00:00:00Z","timestamp":1747958400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,5,23]],"date-time":"2025-05-23T00:00:00Z","timestamp":1747958400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62376286"],"award-info":[{"award-number":["62376286"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62105038"],"award-info":[{"award-number":["62105038"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62105038"],"award-info":[{"award-number":["62105038"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62105038"],"award-info":[{"award-number":["62105038"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Research and Development Program of Beijing Municipal Education Commission","award":["KM202211232001"],"award-info":[{"award-number":["KM202211232001"]}]},{"name":"Research and Development Program of Beijing Municipal Education Commission","award":["KM202211232001"],"award-info":[{"award-number":["KM202211232001"]}]},{"name":"Research and Development Program of Beijing Municipal Education Commission","award":["KM202211232001"],"award-info":[{"award-number":["KM202211232001"]}]},{"name":"Research and Development Program of Beijing Municipal Education Commission","award":["KM202211232001"],"award-info":[{"award-number":["KM202211232001"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2025,9]]},"DOI":"10.1007\/s00371-025-03923-8","type":"journal-article","created":{"date-parts":[[2025,5,23]],"date-time":"2025-05-23T07:31:30Z","timestamp":1747985490000},"page":"9225-9241","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Disentanglement-based compression and generative reconstruction of fixed-scene videos"],"prefix":"10.1007","volume":"41","author":[{"given":"Hongbo","family":"Huang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Longfei","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Meng","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Linkai","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Muye","family":"Sun","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Feng","family":"Gu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guoqing","family":"Hu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhehai","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,23]]},"reference":[{"key":"3923_CR1","unstructured":"Arjovsky, M., Chintala, S., Bottou, L.: Wasserstein generative adversarial networks. In: International Conference on Machine Learning, PMLR, pp. 214\u2013223 (2017)"},{"key":"3923_CR2","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: European Conference on Computer Vision, pp. 213\u2013229. Springer (2020)","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"3923_CR3","unstructured":"Chai, M., Ren, J., Siarohin, A., Tulyakov, S., Woodford, O.: Motion Representations for Articulated Animation. US Patent 11,836,835 (2023)"},{"key":"3923_CR4","doi-asserted-by":"crossref","unstructured":"Chen, H., Gwilliam, M., Lim, S.N., Shrivastava, A.: HNERV: a hybrid neural representation for videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10270\u201310279 (2023)","DOI":"10.1109\/CVPR52729.2023.00990"},{"key":"3923_CR5","first-page":"21557","volume":"34","author":"H Chen","year":"2021","unstructured":"Chen, H., He, B., Wang, H., Ren, Y., Lim, S.N., Shrivastava, A.: NeRV: neural representations for videos. Adv. Neural Inf. Process. Syst. 34, 21557\u201321568 (2021)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"3923_CR6","doi-asserted-by":"crossref","unstructured":"Chen, L.C., Hermans, A., Papandreou, G., Schroff, F., Wang, P., Adam, H.: Masklab: instance segmentation by refining object detection with semantic and direction features. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4013\u20134022 (2018)","DOI":"10.1109\/CVPR.2018.00422"},{"key":"3923_CR7","doi-asserted-by":"crossref","unstructured":"Chen, X., Girshick, R., He, K., Doll\u00e1r, P.: Tensormask: A foundation for dense object segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2061\u20132069 (2019)","DOI":"10.1109\/ICCV.2019.00215"},{"key":"3923_CR8","doi-asserted-by":"crossref","unstructured":"Dai, J., He, K., Sun, J.: Instance-aware semantic segmentation via multi-task network cascades. In: Proceedings of the IEEE conference on Computer Vision and Pattern Recognition, pp. 3150\u20133158 (2016)","DOI":"10.1109\/CVPR.2016.343"},{"key":"3923_CR9","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., et\u00a0al.: An Image is Worth 16 $$\\times $$ 16 words: Transformers for Image Recognition at Scale (2020). arXiv preprint arXiv:2010.11929"},{"key":"3923_CR10","doi-asserted-by":"crossref","unstructured":"Esser, P., Sutter, E., Ommer, B.: A variational u-net for conditional appearance and shape generation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 8857\u20138866 (2018)","DOI":"10.1109\/CVPR.2018.00923"},{"key":"3923_CR11","doi-asserted-by":"crossref","unstructured":"Freeman, A.C., Mayer-Patel, K., Singh, M.: Accelerated Event-Based Feature Detection and Compression for Surveillance Video Systems (2024). arXiv:2312.08213","DOI":"10.1145\/3625468.3647618"},{"key":"3923_CR12","doi-asserted-by":"crossref","unstructured":"Gao, Y., Wei, F., Bao, J., Gu, S., Chen, D., Wen, F., Lian, Z.: High-fidelity and arbitrary face editing. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16115\u201316124 (2021)","DOI":"10.1109\/CVPR46437.2021.01585"},{"key":"3923_CR13","doi-asserted-by":"crossref","unstructured":"Geng, Z., Wang, C., Wei, Y., Liu, Z., Li, H., Hu, H.: Human pose as compositional tokens. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 660\u2013671 (2023)","DOI":"10.1109\/CVPR52729.2023.00071"},{"key":"3923_CR14","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1145\/3422622","volume":"63","author":"I Goodfellow","year":"2020","unstructured":"Goodfellow, I., Pouget-Abadie, J., Mirza, M., Xu, B., Warde-Farley, D., Ozair, S., Courville, A., Bengio, Y.: Generative adversarial networks. Commun. ACM 63, 139\u2013144 (2020)","journal-title":"Commun. ACM"},{"key":"3923_CR15","unstructured":"Gulrajani, I., Ahmed, F., Arjovsky, M., Dumoulin, V., Courville, A.C.: Improved training of Wasserstein GANs. Adv. Neural Inf. Process. Syst. 30 (2017)"},{"key":"3923_CR16","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., Girshick, R.: Mask R-CNN. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2961\u20132969 (2017)","DOI":"10.1109\/ICCV.2017.322"},{"key":"3923_CR17","doi-asserted-by":"crossref","unstructured":"He, S., Liao, W., Yang, M.Y., Yang, Y., Song, Y.Z., Rosenhahn, B., Xiang, T.: Context-aware layout to image generation with enhanced object appearance. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15049\u201315058 (2021)","DOI":"10.1109\/CVPR46437.2021.01480"},{"key":"3923_CR18","doi-asserted-by":"publisher","first-page":"1552","DOI":"10.1109\/TPAMI.2020.3021209","volume":"44","author":"T Hinz","year":"2020","unstructured":"Hinz, T., Heinrich, S., Wermter, S.: Semantic object accuracy for generative text-to-image synthesis. IEEE Trans. Pattern Anal. Mach. Intell. 44, 1552\u20131565 (2020)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"3923_CR19","unstructured":"JTC, I.: Coding of Audio\u2013Visual Objects\u2014Part 2: Visual. ISO\/IEC , 14496-2"},{"key":"3923_CR20","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Mintun, E., Ravi, N., Mao, H., Rolland, C., Gustafson, L., Xiao, T., Whitehead, S., Berg, A.C., Lo, W.Y., et\u00a0al.: Segment Anything (2023). arXiv preprint arXiv:2304.02643","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"3923_CR21","doi-asserted-by":"crossref","unstructured":"Lee, Y., Park, J.: Centermask: Real-time anchor-free instance segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13906\u201313915 (2020)","DOI":"10.1109\/CVPR42600.2020.01392"},{"key":"3923_CR22","doi-asserted-by":"crossref","unstructured":"Li, F., Zhang, H., Liu, S., Guo, J., Ni, L.M., Zhang, L.: DN-DETR: accelerate DETR training by introducing query denoising. IEEE Trans. Pattern Anal. Mach. Intell. 1\u201313 (2023)","DOI":"10.1109\/CVPR52688.2022.01325"},{"key":"3923_CR23","doi-asserted-by":"crossref","unstructured":"Li, F., Zhang, H., Xu, H., Liu, S., Zhang, L., Ni, L.M., Shum, H.Y.: Mask DINO: towards a unified transformer-based framework for object detection and segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3041\u20133050 (2023)","DOI":"10.1109\/CVPR52729.2023.00297"},{"key":"3923_CR24","doi-asserted-by":"publisher","first-page":"163","DOI":"10.1109\/TII.2021.3085669","volume":"18","author":"J Li","year":"2021","unstructured":"Li, J., Chen, J., Sheng, B., Li, P., Yang, P., Feng, D.D., Qi, J.: Automatic detection and classification system of domestic waste via multimodel cascaded convolutional neural network. IEEE Trans. Ind. inform. 18, 163\u2013173 (2021)","journal-title":"IEEE Trans. Ind. inform."},{"key":"3923_CR25","doi-asserted-by":"crossref","unstructured":"Li, J., Li, B., Lu, Y.: Hybrid spatial\u2013temporal entropy modelling for neural video compression. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 1503\u20131511 (2022)","DOI":"10.1145\/3503161.3547845"},{"key":"3923_CR26","doi-asserted-by":"crossref","unstructured":"Li, J., Li, B., Lu, Y.: Neural video compression with diverse contexts. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 22616\u201322626 (2023)","DOI":"10.1109\/CVPR52729.2023.02166"},{"key":"3923_CR27","doi-asserted-by":"crossref","unstructured":"Li, W., Zhang, P., Zhang, L., Huang, Q., He, X., Lyu, S., Gao, J.: Object-driven text-to-image synthesis via adversarial training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12174\u201312182 (2019)","DOI":"10.1109\/CVPR.2019.01245"},{"key":"3923_CR28","doi-asserted-by":"crossref","unstructured":"Li, Y., Qi, H., Dai, J., Ji, X., Wei, Y.: Fully convolutional instance-aware semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2359\u20132367 (2017)","DOI":"10.1109\/CVPR.2017.472"},{"key":"3923_CR29","doi-asserted-by":"crossref","unstructured":"Li, Z., Wang, M., Pi, H., Xu, K., Mei, J., Liu, Y.: E-NERV: expedite neural video representation with disentangled spatial\u2013temporal context. In: European Conference on Computer Vision, pp. 267\u2013284. Springer (2022)","DOI":"10.1007\/978-3-031-19833-5_16"},{"key":"3923_CR30","doi-asserted-by":"publisher","first-page":"50","DOI":"10.1109\/TMM.2021.3120873","volume":"25","author":"X Lin","year":"2021","unstructured":"Lin, X., Sun, S., Huang, W., Sheng, B., Li, P., Feng, D.D.: EAPT: efficient attention pyramid transformer for image processing. IEEE Trans. Multimed. 25, 50\u201361 (2021)","journal-title":"IEEE Trans. Multimed."},{"key":"3923_CR31","unstructured":"Lin, Z., Thekumparampil, K., Fanti, G., Oh, S.: InfoGAN-CR and ModelCentrality: self-supervised model training and selection for disentangling GANS. In: International Conference on Machine Learning. PMLR, pp. 6127\u20136139 (2020)"},{"key":"3923_CR32","doi-asserted-by":"crossref","unstructured":"Liu, B., Chen, Y., Machineni, R.C., Liu, S., Kim, H.S.: MMVC: Learned multi-mode video compression with block-based prediction mode selection and density-adaptive entropy coding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18487\u201318496 (2023)","DOI":"10.1109\/CVPR52729.2023.01773"},{"key":"3923_CR33","doi-asserted-by":"crossref","unstructured":"Liu, X., Liu, W., Mei, T., Ma, H.: A deep learning-based approach to progressive vehicle re-identification for urban surveillance. In: Proceedings of the 14th European Conference on Computer Vision (ECCV 2016), Amsterdam, The Netherlands, October 11\u201314, Part II 14, pp. 869\u2013884. Springer (2016)","DOI":"10.1007\/978-3-319-46475-6_53"},{"key":"3923_CR34","doi-asserted-by":"crossref","unstructured":"Lu, G., Ouyang, W., Xu, D., Zhang, X., Cai, C., Gao, Z.: DVC: an end-to-end deep video compression framework. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11006\u201311015 (2019)","DOI":"10.1109\/CVPR.2019.01126"},{"key":"3923_CR35","unstructured":"Ma, L., Jia, X., Sun, Q., Schiele, B., Tuytelaars, T., Van\u00a0Gool, L.: Pose guided person image generation. Adv. Neural Inf. Process. Syst. 30 (2017)"},{"key":"3923_CR36","doi-asserted-by":"crossref","unstructured":"Meng, D., Chen, X., Fan, Z., Zeng, G., Li, H., Yuan, Y., Sun, L., Wang, J.: Conditional DETR for fast training convergence. In: International Conference on Computer Vision (2021)","DOI":"10.1109\/ICCV48922.2021.00363"},{"key":"3923_CR37","unstructured":"Mirza, M., Osindero, S.: Conditional Generative Adversarial Nets (2014). arXiv preprint arXiv:1411.1784"},{"key":"3923_CR38","unstructured":"Miyato, T., Kataoka, T., Koyama, M., Yoshida, Y.: Spectral Normalization for Generative Adversarial Networks (2018). arXiv preprint arXiv:1802.05957"},{"key":"3923_CR39","doi-asserted-by":"crossref","unstructured":"Newell, A., Yang, K., Deng, J.: Stacked hourglass networks for human pose estimation. In: European Conference on Computer Vision (2016)","DOI":"10.1007\/978-3-319-46484-8_29"},{"key":"3923_CR40","unstructured":"Nowozin, S., Cseke, B., Tomioka, R.: F-GAN: training generative neural samplers using variational divergence minimization. Adv. Neural Inf. Process. Syst. 29 (2016)"},{"key":"3923_CR41","doi-asserted-by":"crossref","unstructured":"Qiao, T., Zhang, J., Xu, D., Tao, D.: MirrorGAN: learning text-to-image generation by redescription. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1505\u20131514 (2019)","DOI":"10.1109\/CVPR.2019.00160"},{"key":"3923_CR42","unstructured":"Roh, B., Shin, J., Shin, W., Kim, S.: Sparsedetr: Efficient end-to-end object detection with learn-able sparsity (2021). arXiv preprint arXiv:2111.14330"},{"key":"3923_CR43","doi-asserted-by":"crossref","unstructured":"Shen, Y., Zhou, B.: Closed-form factorization of latent semantics in GANS. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1532\u20131540 (2021)","DOI":"10.1109\/CVPR46437.2021.00158"},{"key":"3923_CR44","doi-asserted-by":"publisher","first-page":"6662","DOI":"10.1109\/TCYB.2021.3079311","volume":"52","author":"B Sheng","year":"2021","unstructured":"Sheng, B., Li, P., Ali, R., Chen, C.P.: Improving video temporal consistency via broad learning system. IEEE Trans. Cybern. 52, 6662\u20136675 (2021)","journal-title":"IEEE Trans. Cybern."},{"key":"3923_CR45","doi-asserted-by":"crossref","unstructured":"Siarohin, A., Sangineto, E., Lathuiliere, S., Sebe, N.: Deformable GANS for pose-based human image generation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3408\u20133416 (2018)","DOI":"10.1109\/CVPR.2018.00359"},{"key":"3923_CR46","unstructured":"Sochor, J., Jur\u00e1nek, R., \u0160pa\u0148hel, J., Mar\u0161\u00edk, L., \u0160irok\u00fd, A., Herout, A., Zem\u010d\u00edk, P.: Comprehensive Data Set for Automatic Single Camera Visual Speed Measurement (2018). arXiv:1702.06441"},{"key":"3923_CR47","doi-asserted-by":"publisher","first-page":"1649","DOI":"10.1109\/TCSVT.2012.2221191","volume":"22","author":"GJ Sullivan","year":"2012","unstructured":"Sullivan, G.J., Ohm, J.R., Han, W.J., Wiegand, T.: Overview of the high efficiency video coding (HEVC) standard. IEEE Trans. Circuits Syst. Video Technol. 22, 1649\u20131668 (2012)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"3923_CR48","doi-asserted-by":"crossref","unstructured":"Tang, H., Bai, S., Zhang, L., Torr, P.H., Sebe, N.: XingGAN for person image generation. In: Proceedings of the 16th European Conference on Computer Vision (ECCV 2020), Glasgow, UK, August 23\u201328, Part XXV 16, pp. 717\u2013734. Springer (2020)","DOI":"10.1007\/978-3-030-58595-2_43"},{"key":"3923_CR49","doi-asserted-by":"publisher","first-page":"644","DOI":"10.1007\/s11263-022-01722-5","volume":"131","author":"H Tang","year":"2023","unstructured":"Tang, H., Shao, L., Torr, P.H., Sebe, N.: Bipartite graph reasoning GANS for person pose and facial image synthesis. Int. J. Comput. Vis. 131, 644\u2013658 (2023)","journal-title":"Int. J. Comput. Vis."},{"key":"3923_CR50","doi-asserted-by":"crossref","unstructured":"Tompson, J., Goroshin, R., Jain, A., LeCun, Y., Bregler, C.: Efficient object localization using convolutional networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 648\u2013656 (2015)","DOI":"10.1109\/CVPR.2015.7298664"},{"key":"3923_CR51","doi-asserted-by":"crossref","unstructured":"Toshev, A., Szegedy, C.: Deeppose: human pose estimation via deep neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1653\u20131660 (2014)","DOI":"10.1109\/CVPR.2014.214"},{"key":"3923_CR52","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., Polosukhin, I.: Attention is all you need. Adv. Neural Inf. Process. Systems 30 (2017)"},{"key":"3923_CR53","doi-asserted-by":"publisher","first-page":"3349","DOI":"10.1109\/TPAMI.2020.2983686","volume":"43","author":"J Wang","year":"2020","unstructured":"Wang, J., Sun, K., Cheng, T., Jiang, B., Deng, C., Zhao, Y., Liu, D., Mu, Y., Tan, M., Wang, X., et al.: Deep high-resolution representation learning for visual recognition. IEEE Trans. Pattern Anal. Mach. Intell. 43, 3349\u20133364 (2020)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"3923_CR54","doi-asserted-by":"crossref","unstructured":"Wang, X., Kong, T., Shen, C., Jiang, Y., Li, L.: Solo: Segmenting objects by locations. In: Proceedings of the 16th European Conference on Computer Vision (ECCV 2020), Glasgow, UK, August 23\u201328, Part XVIII 16, pp. 649\u2013665. Springer (2020)","DOI":"10.1007\/978-3-030-58523-5_38"},{"key":"3923_CR55","doi-asserted-by":"publisher","first-page":"560","DOI":"10.1109\/TCSVT.2003.815165","volume":"13","author":"T Wiegand","year":"2003","unstructured":"Wiegand, T., Sullivan, G.J., Bjontegaard, G., Luthra, A.: Overview of the H.264\/AVC video coding standard. IEEE Trans. Circuits Syst. Video Technol. 13, 560\u2013576 (2003)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"3923_CR56","doi-asserted-by":"crossref","unstructured":"Wu, C.Y., Singhal, N., Krahenbuhl, P.: Video compression through image interpolation. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 416\u2013431 (2018)","DOI":"10.1007\/978-3-030-01237-3_26"},{"key":"3923_CR57","doi-asserted-by":"crossref","unstructured":"Wu, L., Huang, K., Shen, H., Gao, L.: A Foreground\u2013Background Parallel Compression with Residual Encoding for Surveillance Video (2020). arXiv:2001.06590","DOI":"10.1109\/TCSVT.2020.3027741"},{"key":"3923_CR58","doi-asserted-by":"crossref","unstructured":"Wu, Z., Lischinski, D., Shechtman, E.: Stylespace analysis: disentangled controls for StyleGAN image generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12863\u201312872 (2021)","DOI":"10.1109\/CVPR46437.2021.01267"},{"key":"3923_CR59","first-page":"38571","volume":"35","author":"Y Xu","year":"2022","unstructured":"Xu, Y., Zhang, J., Zhang, Q., Tao, D.: Vitpose: simple vision transformer baselines for human pose estimation. Adv. Neural Inf. Process. Syst. 35, 38571\u201338584 (2022)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"3923_CR60","doi-asserted-by":"crossref","unstructured":"Yang, R., Van\u00a0Gool, L., Timofte, R.: Perceptual Learned Video Compression with Recurrent Conditional GAN (2021). arXiv preprint arXiv:2109.03082 1","DOI":"10.24963\/ijcai.2022\/214"},{"key":"3923_CR61","unstructured":"Zhang, H., Li, F., Liu, S., Zhang, L., Su, H., Zhu, J., Ni, L.M., Shum, H.Y.: DINO: DETR with Improved DeNoising Anchor Boxes for End-to-End Object Detection (2022). arXiv preprint arXiv:2203.03605"},{"key":"3923_CR62","unstructured":"Zhu, X., Su, W., Lu, L., Li, B., Wang, X., Dai, J.: Deformable DETR: deformable transformers for end-to-end object detection. In: International Conference on Learning Representations (2021)"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-03923-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-025-03923-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-03923-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T15:34:28Z","timestamp":1757172868000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-025-03923-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,23]]},"references-count":62,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2025,9]]}},"alternative-id":["3923"],"URL":"https:\/\/doi.org\/10.1007\/s00371-025-03923-8","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,5,23]]},"assertion":[{"value":"14 April 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 May 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"We use public datasets in our experiments.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval and informed consent"}}]}}