{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T18:21:48Z","timestamp":1743013308879,"version":"3.40.3"},"publisher-location":"Cham","reference-count":40,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030289539"},{"type":"electronic","value":"9783030289546"}],"license":[{"start":{"date-parts":[[2019,1,1]],"date-time":"2019-01-01T00:00:00Z","timestamp":1546300800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019]]},"DOI":"10.1007\/978-3-030-28954-6_5","type":"book-chapter","created":{"date-parts":[[2019,9,9]],"date-time":"2019-09-09T19:08:50Z","timestamp":1568056130000},"page":"77-95","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Interpretable Text-to-Image Synthesis with Hierarchical Semantic Layout Generation"],"prefix":"10.1007","author":[{"given":"Seunghoon","family":"Hong","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dingdong","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jongwook","family":"Choi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Honglak","family":"Lee","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,9,10]]},"reference":[{"unstructured":"Banerjee, S., Lavie, A.: METEOR: an automatic metric for MT evaluation with improved correlation with human judgments. In: ACL, pp. 228\u2013231 (2005)","key":"5_CR1"},{"doi-asserted-by":"crossref","unstructured":"Cha, M., Gwon, Y., Kung, H.T.: Adversarial nets with perceptual losses for text-to-image synthesis. In: 2017 IEEE 27th International Workshop on Machine Learning for Signal Processing (MLSP), pp. 1\u20136 (2017)","key":"5_CR2","DOI":"10.1109\/MLSP.2017.8168140"},{"doi-asserted-by":"crossref","unstructured":"Chen, Q., Koltun, V.: Photographic image synthesis with cascaded refinement networks. In: ICCV (2017)","key":"5_CR3","DOI":"10.1109\/ICCV.2017.168"},{"unstructured":"Dash, A., Gamboa, J.C.B., Ahmed, S., Afzal, M.Z., Liwicki, M.: TAC-GAN-Text Conditioned Auxiliary Classifier Generative Adversarial Network. arXiv preprint \narXiv:1703.06412\n\n (2017)","key":"5_CR4"},{"doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: Imagenet: a large-scale hierarchical image database. In: CVPR, pp. 248\u2013255 (2009)","key":"5_CR5","DOI":"10.1109\/CVPR.2009.5206848"},{"doi-asserted-by":"crossref","unstructured":"Dong, H., Yu, S., Wu, C., Guo, Y.: Semantic image synthesis via adversarial learning. In: ICCV, pp. 5707\u20135715 (2017)","key":"5_CR6","DOI":"10.1109\/ICCV.2017.608"},{"doi-asserted-by":"crossref","unstructured":"Dong, H., Zhang, J., McIlwraith, D., Guo, Y.: I2T2I: learning text to image synthesis with textual data augmentation. In: ICIP, pp. 2015\u20132019 (2017)","key":"5_CR7","DOI":"10.1109\/ICIP.2017.8296635"},{"unstructured":"Goodfellow, I.J., et al.: Generative adversarial networks. In: NIPS, pp. 2672\u20132680 (2014)","key":"5_CR8"},{"unstructured":"Ha, D., Eck, D.: A neural representation of sketch drawings. In: ICLR (2018)","key":"5_CR9"},{"issue":"8","key":"5_CR10","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., Schmidhuber, J.: Long short-term memory. Neural Comput. 9(8), 1735\u20131780 (1997)","journal-title":"Neural Comput."},{"doi-asserted-by":"crossref","unstructured":"Isola, P., Zhu, J.Y., Zhou, T., Efros, A.A.: Image-to-image translation with conditional adversarial networks. In: CVPR, pp. 5967\u20135976 (2017)","key":"5_CR11","DOI":"10.1109\/CVPR.2017.632"},{"doi-asserted-by":"crossref","unstructured":"Johnson, J., Alahi, A., Fei-Fei, L.: Perceptual losses for real-time style transfer and super-resolution. In: ECCV, pp. 694\u2013711 (2016)","key":"5_CR12","DOI":"10.1007\/978-3-319-46475-6_43"},{"doi-asserted-by":"crossref","unstructured":"Johnson, J., Gupta, A., Fei-Fei, L.: Image generation from scene graphs. In: CVPR, pp. 1219\u20131228 (2018)","key":"5_CR13","DOI":"10.1109\/CVPR.2018.00133"},{"unstructured":"Karacan, L., Akata, Z., Erdem, A., Erdem, E.: Learning to generate images of outdoor scenes from attributes and semantic layouts. CoRR (2016)","key":"5_CR14"},{"key":"5_CR15","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Computer Vision \u2013 ECCV 2014","author":"Tsung-Yi Lin","year":"2014","unstructured":"Lin, T.Y., et al.: Microsoft COCO: common objects in context. In: ECCV, pp. 740\u2013755 (2014)"},{"unstructured":"Mansimov, E., Parisotto, E., Ba, J.: Generating images from captions with attention. In: ICLR (2016)","key":"5_CR16"},{"unstructured":"Mirza, M., Osindero, S.: Conditional generative adversarial nets. arXiv preprint \narXiv:1411.1784\n\n (2014)","key":"5_CR17"},{"doi-asserted-by":"crossref","unstructured":"Nguyen, A., Yosinski, J., Bengio, Y., Dosovitskiy, A., Clune, J.: Plug & play generative networks: conditional iterative generation of images in latent space. In: CVPR, pp. 3510\u20133520 (2017)","key":"5_CR18","DOI":"10.1109\/CVPR.2017.374"},{"doi-asserted-by":"crossref","unstructured":"Nilsback, M.E., Zisserman, A.: Automated flower classification over a large number of classes. In: Proceedings of the Indian Conference on Computer Vision, Graphics and Image Processing, pp. 722\u2013729 (2008)","key":"5_CR19","DOI":"10.1109\/ICVGIP.2008.47"},{"unstructured":"Odena, A., Olah, C., Shlens, J.: Conditional image synthesis with auxiliary classifier GANs. In: ICML, pp. 2642\u20132651 (2017)","key":"5_CR20"},{"doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.J.: BLEU: a method for automatic evaluation of machine translation. In: ACL, pp. 311\u2013318 (2002)","key":"5_CR21","DOI":"10.3115\/1073083.1073135"},{"doi-asserted-by":"crossref","unstructured":"Reed, S., Akata, Z., Lee, H., Schiele, B.: Learning deep representations of fine-grained visual descriptions. In: CVPR, pp. 49\u201358 (2016)","key":"5_CR22","DOI":"10.1109\/CVPR.2016.13"},{"unstructured":"Reed, S., Akata, Z., Yan, X., Logeswaran, L., Schiele, B., Lee, H.: Generative adversarial text to image synthesis. In: ICML, pp. 1060\u20131069 (2016)","key":"5_CR23"},{"unstructured":"Reed, S., et al.: Parallel multiscale autoregressive density estimation. In: ICML, pp. 2912\u20132921 (2017)","key":"5_CR24"},{"unstructured":"Reed, S.E., Akata, Z., Mohan, S., Tenka, S., Schiele, B., Lee, H.: Learning what and where to draw. In: NIPS, pp. 217\u2013225 (2016)","key":"5_CR25"},{"unstructured":"Salimans, T., Goodfellow, I., Zaremba, W., Cheung, V., Radford, A., Chen, X.: Improved techniques for training GANs. In: NIPS, pp. 2234\u20132242 (2016)","key":"5_CR26"},{"doi-asserted-by":"crossref","unstructured":"Sharma, P., Ding, N., Goodman, S., Soricut, R.: Conceptual captions: a cleaned, hypernymed, image alt-text dataset for automatic image captioning. In: ACL, pp. 2556\u20132565 (2018)","key":"5_CR27","DOI":"10.18653\/v1\/P18-1238"},{"unstructured":"Sharma, S., Suhubdy, D., Michalski, V., Ebrahimi Kahou, S., Bengio, Y.: Chatpainter: improving text to image generation using dialogue. In: ICLR (2018)","key":"5_CR28"},{"unstructured":"Shi, X., Chen, Z., Wang, H., Yeung, D.Y., Wong, W.K., Woo, W.C.: Convolutional LSTM network: a machine learning approach for precipitation nowcasting. In: NIPS, pp. 802\u2013810 (2015)","key":"5_CR29"},{"unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. In: ICLR (2015)","key":"5_CR30"},{"doi-asserted-by":"crossref","unstructured":"Szegedy, C., Vanhoucke, V., Ioffe, S., Shlens, J., Wojna, Z.: Rethinking the inception architecture for computer vision. In: CVPR, pp. 2818\u20132826 (2016)","key":"5_CR31","DOI":"10.1109\/CVPR.2016.308"},{"doi-asserted-by":"crossref","unstructured":"Vedantam, R., Zitnick, C.L., Parikh, D.: CIDEr: consensus-based image description evaluation. In: CVPR, pp. 4566\u20134575 (2015)","key":"5_CR32","DOI":"10.1109\/CVPR.2015.7299087"},{"unstructured":"Villegas, R., Yang, J., Zou, Y., Sohn, S., Lin, X., Lee, H.: Learning to generate long-term future via hierarchical prediction. In: ICML, pp. 3560\u20133569 (2017)","key":"5_CR33"},{"doi-asserted-by":"crossref","unstructured":"Vinyals, O., Toshev, A., Bengio, S., Erhan, D.: Show and tell: a neural image caption generator. In: CVPR, pp. 3156\u20133164 (2015)","key":"5_CR34","DOI":"10.1109\/CVPR.2015.7298935"},{"issue":"8","key":"5_CR35","doi-asserted-by":"publisher","first-page":"4066","DOI":"10.1109\/TIP.2018.2836316","volume":"27","author":"C Wang","year":"2017","unstructured":"Wang, C., Xu, C., Wang, C., Too, D.: Perceptual adversarial networks for image-to-image transformation. IEEE Trans. Image Process. 27(8), 4066\u20134079 (2017)","journal-title":"IEEE Trans. Image Process."},{"key":"5_CR36","doi-asserted-by":"publisher","first-page":"318","DOI":"10.1007\/978-3-319-46493-0_20","volume-title":"Computer Vision \u2013 ECCV 2016","author":"Xiaolong Wang","year":"2016","unstructured":"Wang, X., Gupta, A.: Generative image modeling using style and structure adversarial networks. In: ECCV, pp. 318\u2013335 (2016)"},{"unstructured":"Welinder, P., et al.: Caltech-UCSD Birds 200. Technical Report. CNS-TR-2010-001, California Institute of Technology (2010)","key":"5_CR37"},{"doi-asserted-by":"crossref","unstructured":"Xu, T., et al.: AttnGAN: fine-grained text to image generation with attentional generative adversarial networks. In: CVPR, pp. 1316\u20131324 (2018)","key":"5_CR38","DOI":"10.1109\/CVPR.2018.00143"},{"doi-asserted-by":"crossref","unstructured":"Zhang, H., et al.: StackGAN: text to photo-realistic image synthesis with stacked generative adversarial networks. In: ICCV, pp. 5908\u20135916 (2017)","key":"5_CR39","DOI":"10.1109\/ICCV.2017.629"},{"doi-asserted-by":"crossref","unstructured":"Zhang, Z., Xie, Y., Yang, L.: Photographic text-to-image synthesis with a hierarchically-nested adversarial network. In: CVPR, pp. 1520\u20131529 (2018)","key":"5_CR40","DOI":"10.1109\/CVPR.2018.00649"}],"container-title":["Lecture Notes in Computer Science","Explainable AI: Interpreting, Explaining and Visualizing Deep Learning"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-28954-6_5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,11,13]],"date-time":"2019-11-13T20:12:14Z","timestamp":1573675934000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-030-28954-6_5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019]]},"ISBN":["9783030289539","9783030289546"],"references-count":40,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-28954-6_5","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2019]]},"assertion":[{"value":"10 September 2019","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}