{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T04:22:30Z","timestamp":1784607750773,"version":"3.55.0"},"reference-count":66,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"3","license":[{"start":{"date-parts":[[2023,3,1]],"date-time":"2023-03-01T00:00:00Z","timestamp":1677628800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2023,3,1]],"date-time":"2023-03-01T00:00:00Z","timestamp":1677628800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,3,1]],"date-time":"2023-03-01T00:00:00Z","timestamp":1677628800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U21B2045"],"award-info":[{"award-number":["U21B2045"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U20A20223"],"award-info":[{"award-number":["U20A20223"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Research Innovation Enterprise (RIE) 2020 Industry Alignment Fund Industry Collaboration Projects (IAF-ICP) Funding Initiative"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Circuits Syst. Video Technol."],"published-print":{"date-parts":[[2023,3]]},"DOI":"10.1109\/tcsvt.2022.3210002","type":"journal-article","created":{"date-parts":[[2022,9,26]],"date-time":"2022-09-26T20:46:07Z","timestamp":1664225167000},"page":"1247-1261","source":"Crossref","is-referenced-by-count":11,"title":["Audio-Driven Dubbing for User Generated Contents via Style-Aware Semi-Parametric Synthesis"],"prefix":"10.1109","volume":"33","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0817-2600","authenticated-orcid":false,"given":"Linsen","family":"Song","sequence":"first","affiliation":[{"name":"National Laboratory of Pattern Recognition, CASIA, Center for Research on Intelligent Perception and Computing, CASIA, Center for Excellence in Brain Science and Intelligence Technology, CAS, and the School of Artificial Intelligence, University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wayne","family":"Wu","sequence":"additional","affiliation":[{"name":"SenseTime Research, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0079-7668","authenticated-orcid":false,"given":"Chaoyou","family":"Fu","sequence":"additional","affiliation":[{"name":"National Laboratory of Pattern Recognition, CASIA, Center for Research on Intelligent Perception and Computing, CASIA, Center for Excellence in Brain Science and Intelligence Technology, CAS, and the School of Artificial Intelligence, University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5345-1591","authenticated-orcid":false,"given":"Chen Change","family":"Loy","sequence":"additional","affiliation":[{"name":"S-Laboratory, Nanyang Technological University, Jurong West, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3807-991X","authenticated-orcid":false,"given":"Ran","family":"He","sequence":"additional","affiliation":[{"name":"National Laboratory of Pattern Recognition, CASIA, Center for Research on Intelligent Perception and Computing, CASIA, Center for Excellence in Brain Science and Intelligence Technology, CAS, and the School of Artificial Intelligence, University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","volume-title":"User Generated Content","year":"2020"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1145\/2567948.2576945"},{"key":"ref3","volume-title":"UGC and PGC: Which is the Main Trend","author":"Sun","year":"2017"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TNET.2008.2011358"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/3072959.3073640"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58517-4_42"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.167"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00244"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-47977-5_2"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2015.2502861"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3111648"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3074032"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3106047"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3079897"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3083257"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1145\/3072959.3073699"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/34.927467"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/3072959.3073658"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2017.287"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-483"},{"key":"ref22","first-page":"37","article-title":"End-to-end speech-driven realistic facial animation with temporal GANs","volume-title":"Proc. CVPRW","author":"Vougioukas"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33019299"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-019-01150-y"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1145\/3306346.3323028"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2020.2973374"},{"key":"ref27","article-title":"Everybody\u2019s Talkin\u2019: Let me talk as you want","author":"Song","year":"2020","journal-title":"arXiv:2001.05201"},{"key":"ref28","article-title":"Deep speech: Scaling up end-to-end speech recognition","author":"Hannun","year":"2014","journal-title":"arXiv:1412.5567"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1145\/3414685.3417774"},{"key":"ref30","article-title":"Audio-driven talking face video generation with learning-based personalized head pose","author":"Yi","year":"2020","journal-title":"arXiv:2002.10137"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58545-7_3"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1111\/cgf.12552"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.262"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1145\/3197517.3201283"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1145\/3272127.3275075"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1145\/3355089.3356500"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.244"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2012.2199399"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1145\/1201775.882269"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.537"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1145\/3272127.3275043"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01261-8_41"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.01013"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01246-5_37"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00453"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00813"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2663"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/ICME.2016.7552917"},{"key":"ref49","article-title":"Instance normalization: The missing ingredient for fast stylization","author":"Ulyanov","year":"2016","journal-title":"arXiv:1607.08022"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"ref51","article-title":"Unsupervised representation learning with deep convolutional generative adversarial networks","author":"Radford","year":"2015","journal-title":"arXiv:1511.06434"},{"key":"ref52","first-page":"1","article-title":"Generative adversarial nets","volume-title":"Proc. NeurIPS","author":"Goodfellow"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.304"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00165"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46475-6_43"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0196391"},{"key":"ref57","first-page":"1","article-title":"Adam: A method for stochastic optimization","volume-title":"Proc. ICLR","author":"Kingma"},{"key":"ref58","first-page":"1310","article-title":"On the difficulty of training recurrent neural networks","volume-title":"Proc. ICML","author":"Pascanu"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00227"},{"key":"ref60","first-page":"6629","article-title":"GANs trained by a two time-scale update rule converge to a local nash equilibrium","volume-title":"Proc. NeurIPS","author":"Heusel"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-54427-4_19"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00802"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413532"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2013.249"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1145\/311535.311556"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1145\/1667239.1667251"}],"container-title":["IEEE Transactions on Circuits and Systems for Video Technology"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/76\/10061510\/09903679.pdf?arnumber=9903679","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,22]],"date-time":"2024-01-22T22:46:34Z","timestamp":1705963594000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9903679\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,3]]},"references-count":66,"journal-issue":{"issue":"3"},"URL":"https:\/\/doi.org\/10.1109\/tcsvt.2022.3210002","relation":{},"ISSN":["1051-8215","1558-2205"],"issn-type":[{"value":"1051-8215","type":"print"},{"value":"1558-2205","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,3]]}}}