{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T17:07:31Z","timestamp":1783789651753,"version":"3.55.0"},"reference-count":52,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2021,2,10]],"date-time":"2021-02-10T00:00:00Z","timestamp":1612915200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,2,10]],"date-time":"2021-02-10T00:00:00Z","timestamp":1612915200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100001839","name":"University Grants Committee","doi-asserted-by":"crossref","award":["24206219"],"award-info":[{"award-number":["24206219"]}],"id":[{"id":"10.13039\/501100001839","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100004853","name":"Chinese University of Hong Kong","doi-asserted-by":"crossref","award":["3133233"],"award-info":[{"award-number":["3133233"]}],"id":[{"id":"10.13039\/501100004853","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2021,5]]},"DOI":"10.1007\/s11263-020-01429-5","type":"journal-article","created":{"date-parts":[[2021,2,10]],"date-time":"2021-02-10T07:40:04Z","timestamp":1612942804000},"page":"1451-1466","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":133,"title":["Semantic Hierarchy Emerges in Deep Generative Representations for Scene Synthesis"],"prefix":"10.1007","volume":"129","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1417-1938","authenticated-orcid":false,"given":"Ceyuan","family":"Yang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yujun","family":"Shen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4030-0684","authenticated-orcid":false,"given":"Bolei","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2021,2,10]]},"reference":[{"key":"1429_CR1","doi-asserted-by":"crossref","unstructured":"Abdal, R., Qin, Y., & Wonka, P. (2019). Image2stylegan: How to embed images into the stylegan latent space? In: International conference on computer vision (pp. 4432\u20134441).","DOI":"10.1109\/ICCV.2019.00453"},{"key":"1429_CR2","doi-asserted-by":"crossref","unstructured":"Abdal, R., Qin, Y., & Wonka, P. (2020). Image2stylegan$$++$$: How to edit the embedded images? In: IEEE conference on computer vision and pattern recognition (pp. 8296\u20138305).","DOI":"10.1109\/CVPR42600.2020.00832"},{"key":"1429_CR3","doi-asserted-by":"crossref","unstructured":"Agrawal, P., Girshick, R., & Malik, J. (2014). Analyzing the performance of multilayer neural networks for object recognition. In: European conference on computer vision (pp. 329\u2013344). Springer.","DOI":"10.1007\/978-3-319-10584-0_22"},{"key":"1429_CR4","unstructured":"Alain, G., & Bengio, Y. (2016). Understanding intermediate layers using linear classifier probes. In: International conference on learning representations workshop."},{"key":"1429_CR5","doi-asserted-by":"crossref","unstructured":"Bau, D., Strobelt, H., Peebles, W., Wulff, J., Zhou, B., Zhu, J.-Y., & Torralba, A. (2019). Semantic photo manipulation with a generative image prior. ACM Transactions on Graphics, 38(4), 59.","DOI":"10.1145\/3306346.3323023"},{"key":"1429_CR6","doi-asserted-by":"crossref","unstructured":"Bau, D., Zhou, B., Khosla, A., Oliva, A., & Torralba, A. (2017). Network dissection: Quantifying interpretability of deep visual representations. In: IEEE conference on computer vision and pattern recognition (pp. 6541\u20136549).","DOI":"10.1109\/CVPR.2017.354"},{"key":"1429_CR7","unstructured":"Bau, D., Zhu, J. Y., Strobelt, H., Zhou, B., Tenenbaum, J. B., Freeman, W. T., & Torralba, A. (2018). Gan dissection: Visualizing and understanding generative adversarial networks. In: International conference on learning representations."},{"issue":"8","key":"1429_CR8","doi-asserted-by":"publisher","first-page":"1798","DOI":"10.1109\/TPAMI.2013.50","volume":"35","author":"Y Bengio","year":"2013","unstructured":"Bengio, Y., Courville, A., & Vincent, P. (2013). Representation learning: A review and new perspectives. IEEE Transactions on Pattern Analysis and Machine Intelligence, 35(8), 1798\u20131828.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1429_CR9","unstructured":"Brock, A., Donahue, J., & Simonyan, K. (2018). Large scale gan training for high fidelity natural image synthesis. In: International conference on learning representations."},{"issue":"1","key":"1429_CR10","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/2682628","volume":"34","author":"MM Cheng","year":"2014","unstructured":"Cheng, M. M., Zheng, S., Lin, W. Y., Vineet, V., Sturgess, P., Crook, N., et al. (2014). Imagespirit: Verbal guided image parsing. ACM Transactions on Graphics, 34(1), 1\u201311.","journal-title":"ACM Transactions on Graphics"},{"key":"1429_CR11","doi-asserted-by":"crossref","unstructured":"Choi, Y., Choi, M., Kim, M., Ha, J. W., Kim, S., & Choo, J. (2018). Stargan: Unified generative adversarial networks for multi-domain image-to-image translation. In: IEEE conference on computer vision and pattern recognition (pp. 8789\u20138797).","DOI":"10.1109\/CVPR.2018.00916"},{"key":"1429_CR12","doi-asserted-by":"crossref","unstructured":"Goetschalckx, L., Andonian, A., Oliva, A., & Isola, P. (2019). Ganalyze: Toward visual definitions of cognitive image properties. In: Proceedings of the IEEE international conference on computer vision (pp. 5744\u20135753).","DOI":"10.1109\/ICCV.2019.00584"},{"issue":"5","key":"1429_CR13","doi-asserted-by":"publisher","first-page":"476","DOI":"10.1007\/s11263-017-1048-0","volume":"126","author":"A Gonzalez-Garcia","year":"2018","unstructured":"Gonzalez-Garcia, A., Modolo, D., & Ferrari, V. (2018). Do semantic parts emerge in convolutional neural networks? International Journal of Computer Vision, 126(5), 476\u2013494.","journal-title":"International Journal of Computer Vision"},{"key":"1429_CR14","unstructured":"Goodfellow, I., Pouget-Abadie, J., Mirza, M., Xu, B., Warde-Farley, D., Ozair, S., Courville, A., & Bengio, Y. (2014). Generative adversarial nets. In: Advances in neural information processing systems (pp. 2672\u20132680)."},{"key":"1429_CR15","unstructured":"Heusel, M., Ramsauer, H., Unterthiner, T., Nessler, B., & Hochreiter, S. (2017). Gans trained by a two time-scale update rule converge to a local nash equilibrium. In: Advances in neural information processing systems (pp. 6626\u20136637)."},{"key":"1429_CR16","doi-asserted-by":"crossref","unstructured":"Isola, P., Zhu, J. Y., Zhou, T., & Efros, A. A. (2017). Image-to-image translation with conditional adversarial networks. In: IEEE conference on computer vision and pattern recognition (pp. 1125\u20131134).","DOI":"10.1109\/CVPR.2017.632"},{"key":"1429_CR17","unstructured":"Jahanian, A., Chai, L., & Isola, P. (2019). On the\u201csteerability\u201d of generative adversarial networks. In: International conference on learning representations."},{"key":"1429_CR18","unstructured":"Karacan, L., Akata, Z., Erdem, A., & Erdem, E. (2016) Learning to generate images of outdoor scenes from attributes and semantic layouts. arXiv preprint arXiv:1612.00215."},{"key":"1429_CR19","unstructured":"Karras, T., Aila, T., Laine, S., & Lehtinen, J. (2017). Progressive growing of gans for improved quality, stability, and variation. In: International conference on learning representations."},{"key":"1429_CR20","doi-asserted-by":"crossref","unstructured":"Karras, T., Laine, S., & Aila, T. (2019). A style-based generator architecture for generative adversarial networks. In: IEEE conference on computer vision and pattern recognition (pp. 4401\u20134410).","DOI":"10.1109\/CVPR.2019.00453"},{"issue":"4","key":"1429_CR21","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/2601097.2601101","volume":"33","author":"PY Laffont","year":"2014","unstructured":"Laffont, P. Y., Ren, Z., Tao, X., Qian, C., & Hays, J. (2014). Transient attributes for high-level understanding and editing of outdoor scenes. ACM Transactions on Graphics, 33(4), 1\u201311.","journal-title":"ACM Transactions on Graphics"},{"key":"1429_CR22","doi-asserted-by":"crossref","unstructured":"Liao, J., Yao, Y., Yuan, L., Hua, G., & Kang, S. B. (2017). Visual attribute transfer through deep image analogy. ACM Transactions on Graphics, 36(4), 120.","DOI":"10.1145\/3072959.3073683"},{"key":"1429_CR23","doi-asserted-by":"crossref","unstructured":"Luan, F., Paris, S., Shechtman, E., Bala, K. (2017) Deep photo style transfer. In: IEEE conference on computer vision and pattern recognition (pp. 4990\u20134998).","DOI":"10.1109\/CVPR.2017.740"},{"key":"1429_CR24","doi-asserted-by":"crossref","unstructured":"Mahendran, A., & Vedaldi, A. (2015). Understanding deep image representations by inverting them. In: IEEE conference on computer vision and pattern recognition (pp. 5188\u20135196).","DOI":"10.1109\/CVPR.2015.7299155"},{"key":"1429_CR25","unstructured":"Morcos, A. S., Barrett, D. G., Rabinowitz, N. C., & Botvinick, M. (2018). On the importance of single directions for generalization. In: International conference on learning representations."},{"key":"1429_CR26","unstructured":"Nguyen, A., Dosovitskiy, A., Yosinski, J., Brox, T., & Clune, J. (2016). Synthesizing the preferred inputs for neurons in neural networks via deep generator networks. In: Advances in neural information processing systems (pp. 3387\u20133395)."},{"key":"1429_CR27","doi-asserted-by":"crossref","unstructured":"Nguyen-Phuoc, T., Li, C., Theis, L., Richardt, C., & Yang, Y. L. (2019) Hologan: Unsupervised learning of 3D representations from natural images. In: International conference on computer vision (pp. 7588\u20137597).","DOI":"10.1109\/ICCV.2019.00768"},{"issue":"3","key":"1429_CR28","doi-asserted-by":"publisher","first-page":"145","DOI":"10.1023\/A:1011139631724","volume":"42","author":"A Oliva","year":"2001","unstructured":"Oliva, A., & Torralba, A. (2001). Modeling the shape of the scene: A holistic representation of the spatial envelope. International Journal of Computer Vision, 42(3), 145\u2013175.","journal-title":"International Journal of Computer Vision"},{"key":"1429_CR29","doi-asserted-by":"publisher","first-page":"23","DOI":"10.1016\/S0079-6123(06)55002-2","volume":"155","author":"A Oliva","year":"2006","unstructured":"Oliva, A., & Torralba, A. (2006). Building the gist of a scene: The role of global image features in recognition. Progress in Brain Research, 155, 23\u201336.","journal-title":"Progress in Brain Research"},{"key":"1429_CR30","doi-asserted-by":"crossref","unstructured":"Park, T., Liu, M. Y., Wang, T. C., Zhu, J. Y. (2019). Semantic image synthesis with spatially-adaptive normalization. In: IEEE conference on computer vision and pattern recognition (pp. 2337\u20132346).","DOI":"10.1109\/CVPR.2019.00244"},{"key":"1429_CR31","unstructured":"Park, T., Zhu, J.-Y., Wang, O., Lu, J., Shechtman, E., Efros, A. A., & Zhang, R. (2020). Swapping autoencoder for deep image manipulation. In: Advances in Neural Information Processing Systems."},{"issue":"1\u20132","key":"1429_CR32","doi-asserted-by":"publisher","first-page":"59","DOI":"10.1007\/s11263-013-0695-z","volume":"108","author":"G Patterson","year":"2014","unstructured":"Patterson, G., Xu, C., Su, H., & Hays, J. (2014). The sun attribute database: Beyond categories for deeper scene understanding. International Journal of Computer Vision, 108(1\u20132), 59\u201381.","journal-title":"International Journal of Computer Vision"},{"key":"1429_CR33","unstructured":"Radford, A., Metz, L., & Chintala, S. (2015). Unsupervised representation learning with deep convolutional generative adversarial networks. In: International conference on learning representations."},{"key":"1429_CR34","doi-asserted-by":"crossref","unstructured":"Shaham, T. R., Dekel, T., & Michaeli, T. (2019). Singan: Learning a generative model from a single natural image. In: International conference on computer vision (pp. 4570\u20134580).","DOI":"10.1109\/ICCV.2019.00467"},{"key":"1429_CR35","doi-asserted-by":"crossref","unstructured":"Shen, Y., Gu, J., Tang, X., & Zhou, B. (2020a). Interpreting the latent space of gans for semantic face editing. In: IEEE conference on computer vision and pattern recognition (pp. 9243\u20139252).","DOI":"10.1109\/CVPR42600.2020.00926"},{"key":"1429_CR36","doi-asserted-by":"crossref","unstructured":"Shen, Y., Luo, P., Yan, J., Wang, X., & Tang, X. (2018). Faceid-gan: Learning a symmetry three-player gan for identity-preserving face synthesis. In: IEEE conference on computer vision and pattern recognition (pp. 821\u2013830).","DOI":"10.1109\/CVPR.2018.00092"},{"key":"1429_CR37","doi-asserted-by":"publisher","unstructured":"Shen, Y., Yang, C., Tang, X., & Zhou, B. (2020b). InterFaceGAN: Interpreting the disentangled face representation learned by GANs. IEEE Transactions on Pattern Analysis and Machine Intelligence. https:\/\/doi.org\/10.1109\/TPAMI.2020.3034267.","DOI":"10.1109\/TPAMI.2020.3034267"},{"key":"1429_CR38","unstructured":"Simonyan, K., Vedaldi, A., & Zisserman, A. (2014). Deep inside convolutional networks: Visualising image classification models and saliency maps. In: Workshop at international conference on learning representations."},{"issue":"3","key":"1429_CR39","doi-asserted-by":"publisher","first-page":"391","DOI":"10.1088\/0954-898X_14_3_302","volume":"14","author":"A Torralba","year":"2003","unstructured":"Torralba, A., & Oliva, A. (2003). Statistics of natural image categories. Network: Computation in Neural Systems, 14(3), 391\u2013412.","journal-title":"Network: Computation in Neural Systems"},{"key":"1429_CR40","doi-asserted-by":"crossref","unstructured":"Wang, T. C., Liu, M. Y., Zhu, J. Y., Tao, A., Kautz, J., & Catanzaro, B. (2018). High-resolution image synthesis and semantic manipulation with conditional gans. In: IEEE conference on computer vision and pattern recognition (pp. 8798\u20138807).","DOI":"10.1109\/CVPR.2018.00917"},{"key":"1429_CR41","doi-asserted-by":"crossref","unstructured":"Xiao, J., Hays, J., Ehinger, K. A., Oliva, A., & Torralba A (2010) Sun database: Large-scale scene recognition from abbey to zoo. In: 2010 IEEE computer society conference on computer vision and pattern recognition (pp. 3485\u20133492). IEEE.","DOI":"10.1109\/CVPR.2010.5539970"},{"key":"1429_CR42","doi-asserted-by":"crossref","unstructured":"Xiao, T., Hong, J., & Ma, J. (2018) Elegant: Exchanging latent encodings with gan for transferring multiple face attributes. In: European conference on computer vision (pp. 168\u2013184).","DOI":"10.1007\/978-3-030-01249-6_11"},{"key":"1429_CR43","doi-asserted-by":"crossref","unstructured":"Xiao, T., Liu, Y., Zhou, B., Jiang, Y., & Sun, J. (2018). Unified perceptual parsing for scene understanding. In: Proceedings of the European conference on computer vision (ECCV) (pp. 418\u2013434).","DOI":"10.1007\/978-3-030-01228-1_26"},{"key":"1429_CR44","unstructured":"Yao, S., Hsu, T. M., Zhu, J. Y., Wu, J., Torralba, A., Freeman, B., & Tenenbaum, J. (2018). 3D-aware scene manipulation via inverse graphics. In: Advances in neural information processing systems (pp. 1887\u20131898)."},{"key":"1429_CR45","unstructured":"Yosinski, J., Clune, J., Bengio, Y., & Lipson, H. (2014). How transferable are features in deep neural networks? In: Advances in neural information processing systems (pp. 3320\u20133328)."},{"key":"1429_CR46","unstructured":"Yu, F., Seff, A., Zhang, Y., Song, S., Funkhouser, T., & Xiao, J. (2015) Lsun: Construction of a large-scale image dataset using deep learning with humans in the loop. arXiv preprint arXiv:1506.03365."},{"key":"1429_CR47","doi-asserted-by":"crossref","unstructured":"Zeiler, M. D., & Fergus, R. (2014). Visualizing and understanding convolutional networks. In: European conference on computer vision (pp. 818\u2013833). Springer.","DOI":"10.1007\/978-3-319-10590-1_53"},{"issue":"6","key":"1429_CR48","doi-asserted-by":"publisher","first-page":"2730","DOI":"10.1109\/TCYB.2019.2895837","volume":"50","author":"W Zhang","year":"2019","unstructured":"Zhang, W., Zhang, W., & Gu, J. (2019). Edge-semantic learning strategy for layout estimation in indoor environment. IEEE Transactions on Cybernetics, 50(6), 2730\u20132739.","journal-title":"IEEE Transactions on Cybernetics"},{"key":"1429_CR49","unstructured":"Zhou, B., Khosla, A., Lapedriza, A., Oliva, A., & Torralba, A. (2015). Object detectors emerge in deep scene cnns. In: International conference on learning representations."},{"issue":"6","key":"1429_CR50","doi-asserted-by":"publisher","first-page":"1452","DOI":"10.1109\/TPAMI.2017.2723009","volume":"40","author":"B Zhou","year":"2017","unstructured":"Zhou, B., Lapedriza, A., Khosla, A., Oliva, A., & Torralba, A. (2017). Places: A 10 million image database for scene recognition. IEEE Transactions on Pattern Analysis and Machine Intelligence, 40(6), 1452\u20131464.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1429_CR51","doi-asserted-by":"crossref","unstructured":"Zhu, J., Shen, Y., Zhao, D., & Zhou, B. (2020). In-domain gan inversion for real image editing. In: European conference on computer vision.","DOI":"10.1007\/978-3-030-58520-4_35"},{"key":"1429_CR52","doi-asserted-by":"crossref","unstructured":"Zhu, J. Y., Park, T., Isola, P., & Efros, A. A. (2017). Unpaired image-to-image translation using cycle-consistent adversarial networks. In: International conference on computer vision (pp. 2223\u20132232).","DOI":"10.1109\/ICCV.2017.244"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-020-01429-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-020-01429-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-020-01429-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,5,5]],"date-time":"2021-05-05T18:16:52Z","timestamp":1620238612000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-020-01429-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,2,10]]},"references-count":52,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2021,5]]}},"alternative-id":["1429"],"URL":"https:\/\/doi.org\/10.1007\/s11263-020-01429-5","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,2,10]]},"assertion":[{"value":"31 January 2020","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"31 December 2020","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 February 2021","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}