{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T02:22:09Z","timestamp":1777429329447,"version":"3.51.4"},"reference-count":55,"publisher":"Springer Science and Business Media LLC","issue":"10-11","license":[{"start":{"date-parts":[[2020,5,30]],"date-time":"2020-05-30T00:00:00Z","timestamp":1590796800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,5,30]],"date-time":"2020-05-30T00:00:00Z","timestamp":1590796800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2020,11]]},"DOI":"10.1007\/s11263-020-01325-y","type":"journal-article","created":{"date-parts":[[2020,5,30]],"date-time":"2020-05-30T17:02:31Z","timestamp":1590858151000},"page":"2607-2628","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":13,"title":["Multimodal Image Synthesis with Conditional Implicit Maximum Likelihood Estimation"],"prefix":"10.1007","volume":"128","author":[{"given":"Ke","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shichong","family":"Peng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tianhao","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jitendra","family":"Malik","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,5,30]]},"reference":[{"key":"1325_CR1","unstructured":"Almahairi, A., Rajeswar, S., Sordoni, A., Bachman, P., & Courville, A. (2018). Augmented cyclegan: Learning many-to-many mappings from unpaired data. In ICML."},{"key":"1325_CR2","unstructured":"Arjovsky, M., & Bottou, L. (2017). Towards principled methods for training generative adversarial networks. arXiv:1701.04862."},{"key":"1325_CR3","unstructured":"Arora, S., & Zhang, Y. (2017). Do GANs actually learn the distribution? an empirical study. arXiv:1706.08224."},{"key":"1325_CR4","unstructured":"Bruna, J., Sprechmann, P., & LeCun, Y. (2015). Super-resolution with deep convolutional sufficient statistics. arXiv:1511.05666."},{"key":"1325_CR5","doi-asserted-by":"crossref","unstructured":"Charpiat, G., Hofmann, M., & Sch\u00f6lkopf, B. (2008) Automatic image colorization via multimodal predictions. In European conference on computer vision (pp. 126\u2013139). Springer.","DOI":"10.1007\/978-3-540-88690-7_10"},{"key":"1325_CR6","doi-asserted-by":"crossref","unstructured":"Chen, Q., & Koltun, V. (2017). Photographic image synthesis with cascaded refinement networks. In IEEE international conference on computer vision (ICCV) (Vol. 1, p. 3).","DOI":"10.1109\/ICCV.2017.168"},{"key":"1325_CR7","doi-asserted-by":"crossref","unstructured":"Cordts, M., Omran, M., Ramos, S., Rehfeld, T., Enzweiler, M., Benenson, R., Franke, U., Roth, S., & Schiele, B. (2016). The cityscapes dataset for semantic urban scene understanding. In Proceedings of the IEEE conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2016.350"},{"key":"1325_CR8","doi-asserted-by":"crossref","unstructured":"Dahl, R., Norouzi, M., & Shlens, J. (2017). Pixel recursive super resolution. In 2017 IEEE international conference on computer vision (ICCV) (pp. 5449\u20135458).","DOI":"10.1109\/ICCV.2017.581"},{"key":"1325_CR9","unstructured":"Denton, E. L., Chintala, S., Szlam, A., & Fergus, R. (2015). Deep generative image models using a laplacian pyramid of adversarial networks. In C. Cortes, N. D. Lawrence, D. D. Lee, M. Sugiyama, & R. Garnett (Eds.) Advances in neural information processing systems 28 (pp. 1486\u20131494). Curran Associates, Inc. http:\/\/papers.nips.cc\/paper\/5773-deep-generative-image-models-using-a-laplacian-pyramid-of-adversarial-networks.pdf."},{"key":"1325_CR10","unstructured":"Donahue, J., Kr\u00e4henb\u00fchl, P., & Darrell, T. (2016). Adversarial feature learning. arXiv:1605.09782."},{"key":"1325_CR11","unstructured":"Dumoulin, V., Belghazi, I., Poole, B., Mastropietro, O., Lamb, A., Arjovsky, M., & Courville, A. (2016) Adversarially learned inference. arXiv:1606.00704."},{"key":"1325_CR12","unstructured":"Finn, C., Goodfellow, I., & Levine, S. (2016). Unsupervised learning for physical interaction through video prediction. In Advances in neural information processing systems (pp. 64\u201372)."},{"issue":"5","key":"1325_CR13","first-page":"2","volume":"2014","author":"J Gauthier","year":"2014","unstructured":"Gauthier, J. (2014). Conditional generative adversarial nets for convolutional face generation. Class Project for Stanford CS231N: Convolutional Neural Networks for Visual Recognition, Winter Semester, 2014(5), 2.","journal-title":"Class Project for Stanford CS231N: Convolutional Neural Networks for Visual Recognition, Winter Semester"},{"key":"1325_CR14","doi-asserted-by":"crossref","unstructured":"Ghosh, A., Kulharia, V., Namboodiri, V., Torr, P. H., & Dokania, P. K. (2017). Multi-agent diverse generative adversarial networks. arXiv:1704.02906.","DOI":"10.1109\/CVPR.2018.00888"},{"key":"1325_CR15","unstructured":"Goodfellow, I., Pouget-Abadie, J., Mirza, M., Xu, B., Warde-Farley, D., Ozair, S., Courville, A., & Bengio, Y. (2014). Generative adversarial nets. In Z. Ghahramani, M. Welling, C. Cortes, N. D. Lawrence, & K. Q. Weinberger (Eds.) Advances in neural information processing systems 27 (pp. 2672\u20132680). Curran Associates, Inc. http:\/\/papers.nips.cc\/paper\/5423-generative-adversarial-nets.pdf."},{"key":"1325_CR16","unstructured":"Goodfellow, I. J. (2014). On distinguishability criteria for estimating generative models. arXiv:1412.6515."},{"key":"1325_CR17","unstructured":"Gutmann, M. U., Dutta, R., Kaski, S., & Corander, J. (2014). Likelihood-free inference via classification. arXiv:1407.4981."},{"key":"1325_CR18","unstructured":"Guzman-Rivera, A., Batra, D., & Kohli, P. (2012). Multiple choice learning: Learning to produce multiple structured outputs. In F. Pereira, C. J. C. Burges, L. Bottou, & K. Q. Weinberger (Eds.) Advances in neural information processing systems 25 (pp. 1799\u20131807). Curran Associates, Inc. http:\/\/papers.nips.cc\/paper\/4549-multiple-choice-learning-learning-to-produce-multiple-structured-outputs.pdf."},{"key":"1325_CR19","doi-asserted-by":"crossref","unstructured":"Huang, X., Liu, M. Y., Belongie, S., & Kautz, J. (2018). Multimodal unsupervised image-to-image translation. arXiv:1804.04732.","DOI":"10.1007\/978-3-030-01219-9_11"},{"key":"1325_CR20","doi-asserted-by":"crossref","unstructured":"Isola, P., Zhu, J.-Y., Zhou, T., & Efros, A. A. (2017). Image-to-image translation with conditional adversarial networks. In The IEEE conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2017.632"},{"key":"1325_CR21","doi-asserted-by":"crossref","unstructured":"Johnson, J., Alahi, A., & Fei-Fei, L. (2016). Perceptual losses for real-time style transfer and super-resolution. In European conference on computer vision (pp. 694\u2013711). Springer.","DOI":"10.1007\/978-3-319-46475-6_43"},{"key":"1325_CR22","doi-asserted-by":"crossref","unstructured":"Kaneko, T., Hiramatsu, K., & Kashino, K. (2017). Generative attribute controller with conditional filtered generative adversarial networks. In 2017 IEEE conference on computer vision and pattern recognition (CVPR) (pp. 7006\u20137015). IEEE.","DOI":"10.1109\/CVPR.2017.741"},{"key":"1325_CR23","unstructured":"Karacan, L., Akata, Z., Erdem, A., & Erdem, E. (2016). Learning to generate images of outdoor scenes from attributes and semantic layouts. arXiv:1612.00215."},{"key":"1325_CR24","unstructured":"Larsen, A. B. L., S\u00f8nderby, S. K., Larochelle, H., & Winther, O. (2015). Autoencoding beyond pixels using a learned similarity metric. arXiv:1512.09300."},{"key":"1325_CR25","doi-asserted-by":"crossref","unstructured":"Larsson, G., Maire, M., & Shakhnarovich, G. (2016). Learning representations for automatic colorization. In European conference on computer vision (pp. 577\u2013593). Springer.","DOI":"10.1007\/978-3-319-46493-0_35"},{"key":"1325_CR26","doi-asserted-by":"crossref","unstructured":"Ledig, C., Theis, L., Husz\u00e1r, F., Caballero, J., Cunningham, A., Acosta, A., Aitken, A. P., Tejani, A., Totz, J., Wang, Z., et al. (2017). Photo-realistic single image super-resolution using a generative adversarial network. In CVPR (Vol. 2, p. 4).","DOI":"10.1109\/CVPR.2017.19"},{"key":"1325_CR27","unstructured":"Lee, H. Y., Tseng, H. Y., Mao, Q., Huang, J. B., Lu, Y. D., Singh, M. K., et al. (2018). Drit++: Diverse image-to-image translation via disentangled representations. arXiv:1808.00948."},{"key":"1325_CR28","unstructured":"Lee, S., Ha, J., & Kim, G. (2019). Harmonizing maximum likelihood with gans for multimodal conditional generation. arXiv:1902.09225."},{"key":"1325_CR29","doi-asserted-by":"crossref","unstructured":"Li, C., & Wand, M. (2016). Precomputed real-time texture synthesis with markovian generative adversarial networks. In European conference on computer vision (pp. 702\u2013716). Springer.","DOI":"10.1007\/978-3-319-46487-9_43"},{"key":"1325_CR30","unstructured":"Li, K., & Malik, J. (2016). Fast k-nearest neighbour search via Dynamic Continuous Indexing. In International conference on machine learning (pp. 671\u2013679)."},{"key":"1325_CR31","unstructured":"Li, K., & Malik, J. (2017). Fast k-nearest neighbour search via Prioritized DCI. In International conference on machine learning (pp. 2081\u20132090)."},{"key":"1325_CR32","unstructured":"Li, K., & Malik, J. (2018). Implicit maximum likelihood estimation. arXiv:1809.09087."},{"key":"1325_CR33","unstructured":"Ma, L., Jia, X., Georgoulis, S., Tuytelaars, T., & Gool, L. V. (2018). Exemplar guided unsupervised image-to-image translation with semantic consistency. In ICLR."},{"key":"1325_CR34","unstructured":"Mathieu, M., Couprie, C., & LeCun, Y. (2015). Deep multi-scale video prediction beyond mean square error. arXiv:1511.05440."},{"key":"1325_CR35","unstructured":"Mirza, M., & Osindero, S. (2014). Conditional generative adversarial nets. arXiv:1411.1784."},{"key":"1325_CR36","unstructured":"Oh, J., Guo, X., Lee, H., Lewis, R. L., & Singh, S. (2015). Action-conditional video prediction using deep networks in atari games. In C. Cortes, N. D. Lawrence, D. D. Lee, M. Sugiyama, & R. Garnett (Eds.) Advances in neural information processing systems 28 (pp. 2863\u20132871). Curran Associates, Inc. http:\/\/papers.nips.cc\/paper\/5859-action-conditional-video-prediction-using-deep-networks-in-atari-games.pdf."},{"key":"1325_CR37","doi-asserted-by":"crossref","unstructured":"Pathak, D., Krahenbuhl, P., Donahue, J., Darrell, T., & Efros, A. A. (2016). Context encoders: Feature learning by inpainting. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 2536\u20132544).","DOI":"10.1109\/CVPR.2016.278"},{"key":"1325_CR38","unstructured":"Reed, S. E., Akata, Z., Mohan, S., Tenka, S., Schiele, B., & Lee, H. (2016). Learning what and where to draw. In D. D. Lee, M. Sugiyama, U. V. Luxburg, I. Guyon, & R. Garnett (Eds.). Advances in neural information processing systems 29 (pp. 217\u2013225). Curran Associates, Inc. http:\/\/papers.nips.cc\/paper\/6111-learning-what-and-where-to-draw.pdf."},{"key":"1325_CR39","first-page":"102","volume-title":"European conference on computer vision (ECCV), LNCS","author":"SR Richter","year":"2016","unstructured":"Richter, S. R., Vineet, V., Roth, S., & Koltun, V. (2016). Playing for data: Ground truth from computer games. In B. Leibe, J. Matas, N. Sebe, & M. Welling (Eds.), European conference on computer vision (ECCV), LNCS (pp. 102\u2013118). Berlin: Springer International Publishing."},{"key":"1325_CR40","doi-asserted-by":"crossref","unstructured":"Sangkloy, P., Lu, J., Fang, C., Yu, F., & Hays, J. (2017). Scribbler: Controlling deep image synthesis with sketch and color. In IEEE conference on computer vision and pattern recognition (CVPR) (Vol. 2).","DOI":"10.1109\/CVPR.2017.723"},{"key":"1325_CR41","unstructured":"Simonyan, K., & Zisserman, A. (2014). Very deep convolutional networks for large-scale image recognition. arXiv:1409.1556."},{"key":"1325_CR42","unstructured":"Srivastava, N., Mansimov, E., & Salakhudinov, R. (2015). Unsupervised learning of video representations using lstms. In International conference on machine learning (pp 843\u2013852)."},{"key":"1325_CR43","unstructured":"Vondrick, C., Pirsiavash, H., & Torralba, A. (2016). Generating videos with scene dynamics. In: Advances In neural information processing systems (pp 613\u2013621)."},{"key":"1325_CR44","doi-asserted-by":"crossref","unstructured":"Wang, T. C., Liu, M. Y., Zhu, J. Y., Tao, A., Kautz, J., & Catanzaro, B. (2017). High-resolution image synthesis and semantic manipulation with conditional gans. arXiv:1711.11585.","DOI":"10.1109\/CVPR.2018.00917"},{"key":"1325_CR45","doi-asserted-by":"crossref","unstructured":"Wang, X., & Gupta, A. (2016). Generative image modeling using style and structure adversarial networks. In European conference on computer vision (pp. 318\u2013335). Springer.","DOI":"10.1007\/978-3-319-46493-0_20"},{"key":"1325_CR46","doi-asserted-by":"crossref","unstructured":"Wang, X., Yu, K., Wu, S., Gu, J., Liu, Y., Dong, C., et al. (2018). Esrgan: Enhanced super-resolution generative adversarial networks. arXiv:1809.00219.","DOI":"10.1007\/978-3-030-11021-5_5"},{"key":"1325_CR47","unstructured":"Yang, D., Hong, S., Jang, Y., Zhao, T., & Lee, H. (2019). Diversity-sensitive conditional generative adversarial networks. arXiv:1901.09024."},{"key":"1325_CR48","doi-asserted-by":"crossref","unstructured":"Yoo, D., Kim, N., Park, S., Paek, A. S., & Kweon, I. S. (2016). Pixel-level domain transfer. In European conference on computer vision (pp. 517\u2013532). Springer.","DOI":"10.1007\/978-3-319-46484-8_31"},{"key":"1325_CR49","unstructured":"Yu, F., Xian, W., Chen, Y., Liu, F., Liao, M., Madhavan, V., & Darrell, T. (2018). Bdd100k: A diverse driving video database with scalable annotation tooling. arXiv:1805.04687."},{"key":"1325_CR50","doi-asserted-by":"crossref","unstructured":"Zhang, R., Isola, P., & Efros, A. A. (2016). Colorful image colorization. In European conference on computer vision (pp. 649\u2013666). Springer.","DOI":"10.1007\/978-3-319-46487-9_40"},{"key":"1325_CR51","doi-asserted-by":"crossref","unstructured":"Zhang, R., Isola, P., Efros, A. A., Shechtman, E., & Wang, O. (2018). The unreasonable effectiveness of deep features as a perceptual metric. In The IEEE conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2018.00068"},{"key":"1325_CR52","unstructured":"Zhu, J., Zhang, R., Pathak, D., Darrell, T., Efros, A. A., Wang, O., et al. (2017a). Toward multimodal image-to-image translation. arXiv:1711.11586."},{"key":"1325_CR53","doi-asserted-by":"crossref","unstructured":"Zhu, J. Y., Kr\u00e4henb\u00fchl, P., Shechtman, E., & Efros, A. A. (2016). Generative visual manipulation on the natural image manifold. In European conference on computer vision (pp. 597\u2013613). Springer.","DOI":"10.1007\/978-3-319-46454-1_36"},{"key":"1325_CR54","doi-asserted-by":"crossref","unstructured":"Zhu, J. -Y., Park, T., Isola, P., & Efros, A. A. (2017b). Unpaired image-to-image translation using cycle-consistent adversarial networks. In The IEEE international conference on computer vision (ICCV).","DOI":"10.1109\/ICCV.2017.244"},{"key":"1325_CR55","unstructured":"Zhu, J. Y., Zhang, R., Pathak, D., Darrell, T., Efros, A. A., Wang, O., & Shechtman, E. (2017c). Toward multimodal image-to-image translation. In Advances in neural information processing systems (pp. 465\u2013476)."}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-020-01325-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-020-01325-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-020-01325-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,5,30]],"date-time":"2021-05-30T00:02:56Z","timestamp":1622332976000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-020-01325-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,5,30]]},"references-count":55,"journal-issue":{"issue":"10-11","published-print":{"date-parts":[[2020,11]]}},"alternative-id":["1325"],"URL":"https:\/\/doi.org\/10.1007\/s11263-020-01325-y","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020,5,30]]},"assertion":[{"value":"30 May 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}