{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,5]],"date-time":"2025-12-05T18:54:25Z","timestamp":1764960865221,"version":"3.46.0"},"reference-count":30,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"3","license":[{"start":{"date-parts":[[2025,3,1]],"date-time":"2025-03-01T00:00:00Z","timestamp":1740787200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,3,1]],"date-time":"2025-03-01T00:00:00Z","timestamp":1740787200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,3,1]],"date-time":"2025-03-01T00:00:00Z","timestamp":1740787200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"Major Project for New Generation of AI","award":["2018AAA0100400"],"award-info":[{"award-number":["2018AAA0100400"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61836014","U21B2042","62072457","62006231"],"award-info":[{"award-number":["61836014","U21B2042","62072457","62006231"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"InnoHK Program"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Neural Netw. Learning Syst."],"published-print":{"date-parts":[[2025,3]]},"DOI":"10.1109\/tnnls.2022.3161314","type":"journal-article","created":{"date-parts":[[2022,4,8]],"date-time":"2022-04-08T15:24:33Z","timestamp":1649431473000},"page":"4196-4208","source":"Crossref","is-referenced-by-count":5,"title":["VAG: A Uniform Model for Cross-Modal Visual-Audio Mutual Generation"],"prefix":"10.1109","volume":"36","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7839-7261","authenticated-orcid":false,"given":"Wangli","family":"Hao","sequence":"first","affiliation":[{"name":"National Laboratory of Pattern Recognition (NLPR), Center for Research on Intelligent Perception and Computing (CRIPAC), Institute of Automation, Chinese Academy of Sciences (CASIA), University of Chinese Academy of Sciences (UCAS), Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"He","family":"Guan","sequence":"additional","affiliation":[{"name":"National Laboratory of Pattern Recognition (NLPR), Center for Research on Intelligent Perception and Computing (CRIPAC), Institute of Automation, Chinese Academy of Sciences (CASIA), University of Chinese Academy of Sciences (UCAS), Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2648-3875","authenticated-orcid":false,"given":"Zhaoxiang","family":"Zhang","sequence":"additional","affiliation":[{"name":"Center for Research on Intelligent Perception and Computing, Institute of Automation, University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2014.2308325"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2019.2902489"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2016.2520091"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2013.2267205"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.5555\/2969033.2969125"},{"key":"ref6","article-title":"Conditional generative adversarial nets","author":"Mirza","year":"2014","journal-title":"arXiv:1411.1784"},{"key":"ref7","article-title":"Unsupervised representation learning with deep convolutional generative adversarial networks","author":"Radford","year":"2015","journal-title":"arXiv:1511.06434"},{"key":"ref8","first-page":"214","article-title":"Wasserstein generative adversarial networks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Arjovsky"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.632"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.244"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2019.2907052"},{"key":"ref12","first-page":"1060","article-title":"Generative adversarial text to image synthesis","volume-title":"Proc. 33rd Int. Conf. Int. Conf. Mach. Learn.","volume":"48","author":"Reed"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1145\/3126686.3126723"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.12329"},{"article-title":"Unsupervised and semi-supervised learning with categorical generative adversarial networks","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Springenberg","key":"ref15"},{"key":"ref16","first-page":"2172","article-title":"InfoGAN: Interpretable representation learning by information maximizing generative adversarial nets","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Chen"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-019-01265-2"},{"key":"ref18","first-page":"2172","article-title":"Deep generative image models using a Laplacian pyramid of adversarial networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"28","author":"Denton"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v31i1.10804"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1612.00215"},{"key":"ref21","article-title":"Scribbler: Controlling deep image synthesis with sketch and color","author":"Sangkloy","year":"2016","journal-title":"arXiv:1612.00835"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00917"},{"key":"ref23","first-page":"1152","article-title":"Video-to-video synthesis","volume-title":"Proc. NeurIPS","author":"Wang"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01258-8_18"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00802"},{"key":"ref26","first-page":"1","article-title":"Sound to visual: Hierarchical cross-modal talking face video generation","volume-title":"Proc. IEEE Comput. Soc. Conf. Comput. Vis. Pattern Recognit. Workshops","author":"Chen"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3350986"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6894"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2018.2856090"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/DICTA.2016.7797039"}],"container-title":["IEEE Transactions on Neural Networks and Learning Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/5962385\/10908444\/09753685.pdf?arnumber=9753685","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,5]],"date-time":"2025-12-05T18:39:35Z","timestamp":1764959975000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9753685\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3]]},"references-count":30,"journal-issue":{"issue":"3"},"URL":"https:\/\/doi.org\/10.1109\/tnnls.2022.3161314","relation":{},"ISSN":["2162-237X","2162-2388"],"issn-type":[{"type":"print","value":"2162-237X"},{"type":"electronic","value":"2162-2388"}],"subject":[],"published":{"date-parts":[[2025,3]]}}}