{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,28]],"date-time":"2025-08-28T12:41:45Z","timestamp":1756384905655,"version":"3.37.3"},"reference-count":54,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2021,4,29]],"date-time":"2021-04-29T00:00:00Z","timestamp":1619654400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,4,29]],"date-time":"2021-04-29T00:00:00Z","timestamp":1619654400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2021,7]]},"DOI":"10.1007\/s11263-021-01471-x","type":"journal-article","created":{"date-parts":[[2021,4,29]],"date-time":"2021-04-29T14:03:26Z","timestamp":1619705006000},"page":"2057-2075","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":13,"title":["A Shape-Aware Retargeting Approach to Transfer Human Motion and Appearance in Monocular Videos"],"prefix":"10.1007","volume":"129","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7679-8210","authenticated-orcid":false,"given":"Thiago L.","family":"Gomes","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0053-0004","authenticated-orcid":false,"given":"Renato","family":"Martins","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8093-9880","authenticated-orcid":false,"given":"Jo\u00e3o","family":"Ferreira","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7728-0221","authenticated-orcid":false,"given":"Rafael","family":"Azevedo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6280-5627","authenticated-orcid":false,"given":"Guilherme","family":"Torres","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2973-2232","authenticated-orcid":false,"given":"Erickson R.","family":"Nascimento","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,4,29]]},"reference":[{"key":"1471_CR1","doi-asserted-by":"crossref","unstructured":"Aberman, K., Shi, M., Liao, J., Lischinski, D., Chen, B., & Cohen-Or, D. (2018). Deep video-based performance cloning. CoRR","DOI":"10.1111\/cgf.13632"},{"key":"1471_CR2","doi-asserted-by":"crossref","unstructured":"Aberman, K., Wu, R., Lischinski, D., Chen, B., & Cohen-Or, D. (2019). Learning character-agnostic motion for motion retargeting in 2d. ACM TOG.","DOI":"10.1145\/3306346.3322999"},{"key":"1471_CR3","doi-asserted-by":"crossref","unstructured":"Alldieck, T., Magnor, M., Xu, W., Theobalt, C., & Pons-Moll, G. (2018). Video based reconstruction of 3d people models. In: CVPR.","DOI":"10.1109\/CVPR.2018.00875"},{"key":"1471_CR4","doi-asserted-by":"crossref","unstructured":"Andriluka, M., Pishchulin, L., Gehler, P., & Schiele, B. (2014). 2D human pose estimation: New benchmark and state of the art analysis. In: CVPR.","DOI":"10.1109\/CVPR.2014.471"},{"key":"1471_CR5","doi-asserted-by":"crossref","unstructured":"Anguelov, D., Srinivasan, P., Koller, D., Thrun, S., Rodgers, J., & Davis, J. (2005). Scape: Shape completion and animation of people. ACM Trans Graph.","DOI":"10.1145\/1186822.1073207"},{"key":"1471_CR6","doi-asserted-by":"crossref","unstructured":"Balakrishnan, G., Zhao, A., Dalca, A. V., Durand, F., & Guttag, J. V. (2018). Synthesizing images of humans in unseen poses. In: CVPR.","DOI":"10.1109\/CVPR.2018.00870"},{"key":"1471_CR7","doi-asserted-by":"crossref","unstructured":"Bau, D., Zhu, J. Y., Wulff, J., Peebles, W., Strobelt, H., Zhou, B., & Torralba, A. (2019). Seeing what a gan cannot generate. In: ICCV.","DOI":"10.1109\/ICCV.2019.00460"},{"key":"1471_CR8","doi-asserted-by":"crossref","unstructured":"Bogo, F., Kanazawa, A., Lassner, C., Gehler, P., Romero, J., & Black, M. J. (2016). Keep it SMPL: Automatic estimation of 3D human pose and shape from a single image. In: ECCV.","DOI":"10.1007\/978-3-319-46454-1_34"},{"key":"1471_CR9","doi-asserted-by":"crossref","unstructured":"Cao, Z., Simon, T., Wei, S. E., & Sheikh, Y. (2017). Realtime multi-person 2d pose estimation using part affinity fields. In: CVPR.","DOI":"10.1109\/CVPR.2017.143"},{"key":"1471_CR10","doi-asserted-by":"crossref","unstructured":"Chan, C., Ginosar, S., Zhou, T., & Efros, A. (2019). Everybody dance now. In: ICCV.","DOI":"10.1109\/ICCV.2019.00603"},{"key":"1471_CR11","doi-asserted-by":"crossref","unstructured":"Choi, K. J., & Ko, H. S. (2000). On-line motion retargeting. Journal of Visualization and Computer Animation.","DOI":"10.1002\/1099-1778(200012)11:5<223::AID-VIS236>3.0.CO;2-5"},{"key":"1471_CR12","doi-asserted-by":"crossref","unstructured":"Criminisi, A., Perez, P., & Toyama, K. (2004). Region filling and object removal by exemplar-based image inpainting. IEEE TIP.","DOI":"10.1109\/TIP.2004.833105"},{"key":"1471_CR13","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4612-6333-3","volume-title":"A practical guide to splines","author":"C DeBoor","year":"1978","unstructured":"DeBoor, C., DeBoor, C., Math\u00e9maticien, E. U., DeBoor, C., & DeBoor, C. (1978). A practical guide to splines (Vol. 27). Berlin: Springer."},{"key":"1471_CR14","doi-asserted-by":"crossref","unstructured":"Dosovitskiy, A., Springenberg, J. T., & Brox, T. (2015). Learning to generate chairs with convolutional neural networks. In: CVPR.","DOI":"10.1109\/CVPR.2015.7298761"},{"key":"1471_CR15","doi-asserted-by":"crossref","unstructured":"Esser, P., Sutter, E., & Ommer, B. (2018). A variational u-net for conditional appearance and shape generation. In: CVPR.","DOI":"10.1109\/CVPR.2018.00923"},{"key":"1471_CR16","doi-asserted-by":"crossref","unstructured":"Gleicher, M. (1998). Retargetting motion to new characters. In: SIGGRAPH.","DOI":"10.1145\/280814.280820"},{"key":"1471_CR17","doi-asserted-by":"crossref","unstructured":"Gomes, T., Martins, R., Ferreira, J., & Nascimento, E. (2020). Do as I do: Transferring human motion and appearance between monocular videos with spatial and temporal constraints. In: WACV.","DOI":"10.1109\/WACV45572.2020.9093395"},{"key":"1471_CR18","doi-asserted-by":"crossref","unstructured":"Gong, K., Liang, X., Li, Y., Chen, Y., Yang, M., & Lin, L. (2018). Instance-level human parsing via part grouping network. In: ECCV.","DOI":"10.1007\/978-3-030-01225-0_47"},{"key":"1471_CR19","doi-asserted-by":"crossref","unstructured":"Hassan, M., Choutas, V., Tzionas, D., & Black, M.J. (2019). Resolving 3D human pose ambiguities with 3D scene constraints. In: ICCV.","DOI":"10.1109\/ICCV.2019.00237"},{"key":"1471_CR20","doi-asserted-by":"crossref","unstructured":"Kanazawa, A., Black, M.J., Jacobs, D.W., & Malik, J. (2018). End-to-end recovery of human shape and pose. In: CVPR.","DOI":"10.1109\/CVPR.2018.00744"},{"key":"1471_CR21","doi-asserted-by":"crossref","unstructured":"Kolotouros, N., Pavlakos, G., Black, M.J., & Daniilidis, K. (2019). Learning to reconstruct 3d human pose and shape via model-fitting in the loop. In: ICCV.","DOI":"10.1109\/ICCV.2019.00234"},{"key":"1471_CR22","doi-asserted-by":"crossref","unstructured":"Lassner, C., Pons-Moll, G., & Gehler, P. V. (2017a) A generative model for people in clothing. In: ICCV.","DOI":"10.1109\/ICCV.2017.98"},{"key":"1471_CR23","doi-asserted-by":"crossref","unstructured":"Lassner, C., Romero, J., Kiefel, M., Bogo, F., Black, M. J., & Gehler, P. V. (2017b) Unite the people: Closing the loop between 3d and 2d human representations. In: CVPR.","DOI":"10.1109\/CVPR.2017.500"},{"key":"1471_CR24","doi-asserted-by":"crossref","unstructured":"Levi, Z., & Gotsman, C. (2015). Smooth rotation enhanced as-rigid-as-possible mesh animation. T-VCG.","DOI":"10.1109\/TVCG.2014.2359463"},{"key":"1471_CR25","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., & Zitnick, C.L. (2014) Microsoft coco: Common objects in context. In: ECCV.","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"1471_CR26","doi-asserted-by":"crossref","unstructured":"Liu, W., Piao, Z., Jie, M., Luo, W., Ma, L., & Gao, S. (2019) Liquid warping GAN: A unified framework for human motion imitation, appearance transfer and novel view synthesis. In: ICCV.","DOI":"10.1109\/ICCV.2019.00600"},{"key":"1471_CR27","doi-asserted-by":"crossref","unstructured":"Loper, M., Mahmood, N., Romero, J., Pons-Moll, G., & Black, M.J. (2015) Smpl: A skinned multi-person linear model. ACM Trans Graph.","DOI":"10.1145\/2816795.2818013"},{"key":"1471_CR28","doi-asserted-by":"crossref","unstructured":"Ma, L., Jia, X., Sun, Q., Schiele, B., Tuytelaars, T., & Van\u00a0Gool, L. (2017) Pose guided person image generation. In: NIPS.","DOI":"10.1109\/CVPR.2018.00018"},{"key":"1471_CR29","doi-asserted-by":"crossref","unstructured":"Mahmood, N., Ghorbani, N., Troje, N. F., Pons-Moll, G., & Black, M. J. (2019) AMASS: Archive of motion capture as surface shapes. In: ICCV.","DOI":"10.1109\/ICCV.2019.00554"},{"key":"1471_CR30","doi-asserted-by":"crossref","unstructured":"Marra, F., Gragnaniello, D., Verdoliva, L., & Poggi, G. (2020) A full-image full-resolution end-to-end-trainable cnn framework for image forgery detection. IEEE Access.","DOI":"10.1109\/ACCESS.2020.3009877"},{"key":"1471_CR31","doi-asserted-by":"crossref","unstructured":"Mehta, D., Rhodin, H., Casas, D., Fua, P., Sotnychenko, O., Xu, W., & Theobalt, C. (2017) Monocular 3d human pose estimation in the wild using improved cnn supervision. In: 3DV.","DOI":"10.1109\/3DV.2017.00064"},{"key":"1471_CR32","doi-asserted-by":"crossref","unstructured":"Mir, A., Alldieck, T., & Pons-Moll, G. (2020) Learning to transfer texture from clothing images to 3d humans. In: CVPR, IEEE.","DOI":"10.1109\/CVPR42600.2020.00705"},{"key":"1471_CR33","doi-asserted-by":"crossref","unstructured":"Neverova, N., G\u00fcler, R. A., & Kokkinos, I. (2018) Dense pose transfer. In: ECCV.","DOI":"10.1007\/978-3-030-01219-9_8"},{"key":"1471_CR34","doi-asserted-by":"crossref","unstructured":"Peng, X. B., Kanazawa, A., Malik, J., Abbeel, P., & Levine, S. (2018) Sfv: Reinforcement learning of physical skills from videos. ACM Trans Graph.","DOI":"10.1145\/3272127.3275014"},{"key":"1471_CR35","doi-asserted-by":"crossref","unstructured":"Shysheya, A., Zakharov, E., Aliev, K. A., Bashirov, R., Burkov, E., Iskakov, K., Ivakhnenko, A., Malkov, Y., Pasechnik, I., Ulyanov, D., Vakhitov, A., & Lempitsky, V. (2019) Textured neural avatars. In: CVPR.","DOI":"10.1109\/CVPR.2019.00249"},{"key":"1471_CR36","unstructured":"Sigal, L., Balan, A., & Black, M. J. (2007) Combined discriminative and generative articulated pose and non-rigid shape estimation. In: NIPS."},{"key":"1471_CR37","doi-asserted-by":"crossref","unstructured":"Simon, T., Joo, H., Matthews, I., & Sheikh, Y. (2017) Hand keypoint detection in single images using multiview bootstrapping. In: CVPR.","DOI":"10.1109\/CVPR.2017.494"},{"key":"1471_CR38","unstructured":"Sun, Y. T., Fu, Q. C., Jiang, Y. R., Liu, Z., Lai, Y. K., Fu, H., & Gao, L. (2020) Human motion transfer with 3d constraints and detail enhancement. arXiv:2003.13510."},{"key":"1471_CR39","doi-asserted-by":"crossref","unstructured":"Tatarchenko, M., Dosovitskiy, A., & Brox, T. (2015) Single-view to multi-view: Reconstructing unseen views with a convolutional network. CoRR.","DOI":"10.1007\/978-3-319-46478-7_20"},{"issue":"2","key":"1471_CR40","doi-asserted-by":"publisher","first-page":"701","DOI":"10.1111\/cgf.14022","volume":"39","author":"A Tewari","year":"2020","unstructured":"Tewari, A., Fried, O., Thies, J., Sitzmann, V., Lombardi, S., Sunkavalli, K., et al. (2020). State of the art on neural rendering. Computer Graphics Forum, 39(2), 701\u2013727. https:\/\/doi.org\/10.1111\/cgf.14022.","journal-title":"Computer Graphics Forum"},{"key":"1471_CR41","unstructured":"Unterthiner, T., van Steenkiste, S., Kurach, K., Marinier, R., Michalski, M., & Gelly, S. (2019). Towards accurate generative models of video: A new metric & challenges. arXiv:1812.01717"},{"key":"1471_CR42","doi-asserted-by":"crossref","unstructured":"Villegas, R., Yang, J., Ceylan, D., & Lee, H. (2018) Neural kinematic networks for unsupervised motion retargetting. In: CVPR.","DOI":"10.1109\/CVPR.2018.00901"},{"key":"1471_CR43","doi-asserted-by":"crossref","unstructured":"Wang, C., Huang, H., Han, X., & Wang, J. (2019) Video inpainting by jointly learning temporal structure and spatial details. In: AAAI.","DOI":"10.1609\/aaai.v33i01.33015232"},{"key":"1471_CR44","doi-asserted-by":"crossref","unstructured":"Wang, S., Wang, O., Zhang, R., Owens, A., & Efros, A.A. (2020) Cnn-generated images are surprisingly easy to spot... for now. In: CVPR.","DOI":"10.1109\/CVPR42600.2020.00872"},{"key":"1471_CR45","doi-asserted-by":"crossref","unstructured":"Wang, S. Y., Wang, O., Zhang, R., Owens, A., & Efros, A. A. (2020) Cnn-generated images are surprisingly easy to spot... for now. In: CVPR.","DOI":"10.1109\/CVPR42600.2020.00872"},{"key":"1471_CR46","unstructured":"Wang, T. C., Liu, M. Y., Zhu, J. Y., Liu, G., Tao, A., Kautz, J., & Catanzaro, B. (2018) Video-to-video synthesis. In: NIPS."},{"key":"1471_CR47","doi-asserted-by":"crossref","unstructured":"Wang, Z., Bovik, A. C., Sheikh, H. R., & Simoncelli, E. P. (2004) Image quality assessment: From error visibility to structural similarity. IEEE TIP.","DOI":"10.1109\/TIP.2003.819861"},{"key":"1471_CR48","doi-asserted-by":"crossref","unstructured":"Wei, S. E., Ramakrishna, V., Kanade, T., & Sheikh, Y. (2016) Convolutional pose machines. In: CVPR.","DOI":"10.1109\/CVPR.2016.511"},{"key":"1471_CR49","doi-asserted-by":"crossref","unstructured":"Xu, R., Li, X., Zhou, B., & Loy, C. C. (2019) Deep flow-guided video inpainting. In: CVPR.","DOI":"10.1109\/CVPR.2019.00384"},{"key":"1471_CR50","unstructured":"Yang, J., Reed, S., Yang, M. H., & Lee, H. (2015) Weakly-supervised disentangling with recurrent transformations for 3d view synthesis. In: NIPS."},{"key":"1471_CR51","doi-asserted-by":"crossref","unstructured":"Yu, J., Lin, Z., Yang, J., Shen, X., Lu, X., & Huang, T. S. (2018) Generative image inpainting with contextual attention. In: CVPR.","DOI":"10.1109\/CVPR.2018.00577"},{"key":"1471_CR52","doi-asserted-by":"crossref","unstructured":"Yu, J., Lin, Z., Yang, J., Shen, X., Lu, X., & Huang, T. (2019) Free-form image inpainting with gated convolution. In: ICCV.","DOI":"10.1109\/ICCV.2019.00457"},{"key":"1471_CR53","doi-asserted-by":"crossref","unstructured":"Zhang, R., Isola, P., Efros, A. A., Shechtman, E., & Wang, O. (2018) The unreasonable effectiveness of deep features as a perceptual metric. In: CVPR.","DOI":"10.1109\/CVPR.2018.00068"},{"key":"1471_CR54","doi-asserted-by":"crossref","unstructured":"Zhao, B., Wu, X., Cheng, Z., Liu, H., & Feng, J. (2017) Multi-view image generation from a single-view. CoRR.","DOI":"10.1145\/3240508.3240536"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-021-01471-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-021-01471-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-021-01471-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,29]],"date-time":"2024-08-29T21:01:08Z","timestamp":1724965268000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-021-01471-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,4,29]]},"references-count":54,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2021,7]]}},"alternative-id":["1471"],"URL":"https:\/\/doi.org\/10.1007\/s11263-021-01471-x","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"type":"print","value":"0920-5691"},{"type":"electronic","value":"1573-1405"}],"subject":[],"published":{"date-parts":[[2021,4,29]]},"assertion":[{"value":"16 September 2020","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 April 2021","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 April 2021","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}