{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,18]],"date-time":"2025-10-18T20:59:25Z","timestamp":1760821165821},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2020,5,1]],"date-time":"2020-05-01T00:00:00Z","timestamp":1588291200000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,5,1]],"date-time":"2020-05-01T00:00:00Z","timestamp":1588291200000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J. Comput. Sci. Technol."],"published-print":{"date-parts":[[2020,5]]},"DOI":"10.1007\/s11390-020-0305-9","type":"journal-article","created":{"date-parts":[[2020,6,8]],"date-time":"2020-06-08T21:02:32Z","timestamp":1591650152000},"page":"522-537","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":14,"title":["A Comprehensive Pipeline for Complex Text-to-Image Synthesis"],"prefix":"10.1007","volume":"35","author":[{"given":"Fei","family":"Fang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fei","family":"Luo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hong-Pan","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hua-Jian","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alix L. H.","family":"Chow","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chun-Xia","family":"Xiao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,5,29]]},"reference":[{"key":"305_CR1","doi-asserted-by":"crossref","unstructured":"Lin T Y, Maire M, Belongie S et al. Microsoft COCO: Common objects in context. In Proc. the 13th European Conference on Computer Vision, September 2014, pp.740-755.","DOI":"10.1007\/978-3-319-10602-1_48"},{"issue":"1","key":"305_CR2","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","volume":"123","author":"R Krishna","year":"2017","unstructured":"Krishna R, Zhu Y, Groth O et al. Visual genome: Connecting language and vision using crowdsourced dense image annotations. International Journal of Computer Vision, 2017, 123(1): 32-73.","journal-title":"International Journal of Computer Vision"},{"key":"305_CR3","unstructured":"Mansimov E, Parisotto E, Ba J L et al. Generating images from captions with attention. arXiv:1511.02793, 2015. \nhttps:\/\/arxiv.org\/abs\/1511.02793\n\n, October 2019."},{"key":"305_CR4","unstructured":"Reed S, Akata Z, Yan X et al. Generative adversarial text to image synthesis. arXiv:1605.05396, 2016. \nhttps:\/\/arxiv.org\/abs\/1605.05396\n\n, October 2019."},{"key":"305_CR5","doi-asserted-by":"crossref","unstructured":"Zhang H, Xu T, Li H et al. StackGAN: Text to photorealistic image synthesis with stacked generative adversarial networks. In Proc. the 2017 IEEE International Conference on Computer Vision, October 2017, pp.5907-5915.","DOI":"10.1109\/ICCV.2017.629"},{"key":"305_CR6","doi-asserted-by":"crossref","unstructured":"Lalonde J F, Hoiem D, Efros A A et al. Photo clip art. ACM Transactions on Graphics, 2007, 26(3): Article No. 3.","DOI":"10.1145\/1276377.1276381"},{"key":"305_CR7","doi-asserted-by":"crossref","unstructured":"Chen T, Cheng M M, Tan P et al. Sketch2Photo: Internet image montage. ACM Transactions on Graphics, 2009, 28(5): Article No. 124.","DOI":"10.1145\/1618452.1618470"},{"issue":"5","key":"305_CR8","doi-asserted-by":"publisher","first-page":"824","DOI":"10.1109\/TVCG.2012.148","volume":"19","author":"T Chen","year":"2013","unstructured":"Chen T, Tan P, Ma L Q et al. PoseShop: Human image database construction and personalized content synthesis. IEEE Transactions on Visualization and Computer Graphics, 2013, 19(5): 824-837.","journal-title":"IEEE Transactions on Visualization and Computer Graphics"},{"issue":"9","key":"305_CR9","doi-asserted-by":"publisher","first-page":"2559","DOI":"10.1109\/TVCG.2017.2759265","volume":"24","author":"F Fang","year":"2018","unstructured":"Fang F, Yi M, Feng H et al. Narrative collage of image collections by scene graph recombination. IEEE Transactions on Visualization and Computer Graphics, 2018, 24(9): 2559-2572.","journal-title":"IEEE Transactions on Visualization and Computer Graphics"},{"key":"305_CR10","doi-asserted-by":"crossref","unstructured":"Zitnick C L, Parikh D. Bringing semantics into focus using visual abstraction. In Proc. the IEEE Conference on Computer Vision and Pattern Recognition, June 2013, pp.3009-3016.","DOI":"10.1109\/CVPR.2013.387"},{"key":"305_CR11","doi-asserted-by":"crossref","unstructured":"Zitnick C L, Parikh D, Vanderwende L. Learning the visual interpretation of sentences. In Proc. the IEEE International Conference on Computer Vision, December 2013, pp.1681-1688.","DOI":"10.1109\/ICCV.2013.211"},{"key":"305_CR12","doi-asserted-by":"crossref","unstructured":"Coyne B, Sproat R. WordsEye: An automatic text-to-scene conversion system. In Proc. the 28th Annual Conference on Computer Graphics and Interactive Techniques, August 2001, pp.487-496.","DOI":"10.1145\/383259.383316"},{"key":"305_CR13","doi-asserted-by":"crossref","unstructured":"Chang A, Savva M, Manning C D. Learning spatial knowledge for text to 3D scene generation. In Proc. the 2014 Conference on Empirical Methods in Natural Language Processing, October 2014, pp.2028-2038.","DOI":"10.3115\/v1\/D14-1217"},{"key":"305_CR14","unstructured":"Reed S, van den Oord A, Kalchbrenner N et al. Generating interpretable images with controllable structure. In Proc. the International Conference on Learning Representations, April 2017."},{"key":"305_CR15","unstructured":"Goodfellow I, Pouget-Abadie J, Mirza M et al. Generative adversarial nets. In Proc. the Annual Conference on Neural Information Processing Systems, December 2014, pp.2672-2680."},{"key":"305_CR16","unstructured":"Reed S E, Akata Z, Mohan S et al. Learning what and where to draw. In Proc. the Annual Conference on Neural Information Processing Systems, December 2016, pp.217-225."},{"issue":"8","key":"305_CR17","doi-asserted-by":"publisher","first-page":"1947","DOI":"10.1109\/TPAMI.2018.2856256","volume":"41","author":"H Zhang","year":"2019","unstructured":"Zhang H, Xu T, Li H et al. StackGAN++: Realistic image synthesis with stacked generative adversarial networks. IEEE Transactions on Pattern Analysis and Machine Intelligence, 2019, 41(8): 1947-1962.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"305_CR18","doi-asserted-by":"crossref","unstructured":"Xu T, Zhang P, Huang Q et al. AttnGAN: Fine-grained text to image generation with attentional generative adversarial networks. In Proc. the IEEE Conference on Computer Vision and Pattern Recognition, June 2018, pp.1316-1324.","DOI":"10.1109\/CVPR.2018.00143"},{"key":"305_CR19","doi-asserted-by":"crossref","unstructured":"Yin G, Liu B, Sheng L et al. Semantics disentangling for text-to-image generation. In Proc. the IEEE Conference on Computer Vision and Pattern Recognition, June 2019, pp.2327-2336.","DOI":"10.1109\/CVPR.2019.00243"},{"key":"305_CR20","doi-asserted-by":"crossref","unstructured":"Zhou X, Huang S, Li B et al. Text guided person image synthesis. In Proc. the IEEE Conference on Computer Vision and Pattern Recognition, June 2019, pp.3663-3672.","DOI":"10.1109\/CVPR.2019.00378"},{"key":"305_CR21","doi-asserted-by":"crossref","unstructured":"Tan H, Liu X, Li X et al. Semantics-enhanced adversarial nets for text-to-image synthesis. In Proc. the IEEE International Conference on Computer Vision, October 2019, pp.10500-10509.","DOI":"10.1109\/ICCV.2019.01060"},{"key":"305_CR22","doi-asserted-by":"crossref","unstructured":"Qiao T, Zhang J, Xu D et al. MirrorGAN: Learning text-to-image generation by redescription. In Proc. the IEEE Conference on Computer Vision and Pattern Recognition, June 2019, pp.1505-1514.","DOI":"10.1109\/CVPR.2019.00160"},{"key":"305_CR23","doi-asserted-by":"crossref","unstructured":"Johnson J, Gupta A, Li F F. Image generation from scene graphs. In Proc. the IEEE Conference on Computer Vision and Pattern Recognition, June 2018, pp.1219-1228.","DOI":"10.1109\/CVPR.2018.00133"},{"key":"305_CR24","doi-asserted-by":"crossref","unstructured":"Li W, Zhang P, Zhang L et al. Object-driven text-to-image synthesis via adversarial training. In Proc. the IEEE Conference on Computer Vision and Pattern Recognition, June 2019, pp.12174-12182.","DOI":"10.1109\/CVPR.2019.01245"},{"key":"305_CR25","unstructured":"Hinz T, Heinrich S, Wermter S. Generating multiple objects at spatially distinct locations. arXiv:1901.00686, 2019. \nhttps:\/\/arxiv.org\/abs\/1901.00686\n\n, October 2019."},{"key":"305_CR26","unstructured":"Xu K, Ba J, Kiros R et al. Show, attend and tell: Neural image caption generation with visual attention. In Proc. the 32nd International Conference on Machine Learning, July 2015, pp.2048-2057."},{"key":"305_CR27","doi-asserted-by":"crossref","unstructured":"Karpathy A, Li F F. Deep visual-semantic alignments for generating image descriptions. In Proc. the IEEE Conference on Computer Vision and Pattern Recognition, June 2015, pp.3128-3137.","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"305_CR28","doi-asserted-by":"crossref","unstructured":"Johnson J, Karpathy A, Li F F. DenseCap: Fully convolutional localization networks for dense captioning. In Proc. the 2016 IEEE Conference on Computer Vision and Pattern Recognition, June 2016, pp.4565-4574.","DOI":"10.1109\/CVPR.2016.494"},{"key":"305_CR29","doi-asserted-by":"crossref","unstructured":"Krause J, Johnson J, Krishna R et al. A hierarchical approach for generating descriptive image paragraphs. In Proc. the 2017 IEEE Conference on Computer Vision and Pattern Recognition, July 2017, pp.3337-3345.","DOI":"10.1109\/CVPR.2017.356"},{"key":"305_CR30","doi-asserted-by":"crossref","unstructured":"Yao L, Torabi A, Cho K et al. Describing videos by exploiting temporal structure. In Proc. the IEEE International Conference on Computer Vision, December 2015, pp.4507-4515.","DOI":"10.1109\/ICCV.2015.512"},{"key":"305_CR31","doi-asserted-by":"crossref","unstructured":"Yu H, Wang J, Huang Z et al. Video paragraph captioning using hierarchical recurrent neural networks. In Proc. the 2015 IEEE Conference on Computer Vision and Pattern Recognition, June 2016, pp.4584-4593.","DOI":"10.1109\/CVPR.2016.496"},{"key":"305_CR32","doi-asserted-by":"crossref","unstructured":"Li A, Sun J, Ng J Y H et al. Generating holistic 3D scene abstractions for text-based image retrieval. In Proc. the IEEE Conference on Computer Vision and Pattern Recognition, July 2017, pp.1942-1950.","DOI":"10.1109\/CVPR.2017.210"},{"key":"305_CR33","doi-asserted-by":"crossref","unstructured":"Fellbaum C. WordNet. In Theory and Applications of Ontology: Computer Applications, Poli P, Healy M, Kameas A (eds.), Springer Netherlands, 2010, pp.231-243.","DOI":"10.1007\/978-90-481-8847-5_10"},{"key":"305_CR34","doi-asserted-by":"crossref","unstructured":"He K, Gkioxari G, Doll\u00e1r P et al. Mask R-CNN. In Proc. the IEEE International Conference on Computer Vision, October 2017, pp.2980-2988.","DOI":"10.1109\/ICCV.2017.322"},{"key":"305_CR35","doi-asserted-by":"crossref","unstructured":"Laina I, Rupprecht C, Belagiannis V et al. Deeper depth prediction with fully convolutional residual networks. In Proc. the 4th International Conference on 3D Vision, October 2016, pp.239-248.","DOI":"10.1109\/3DV.2016.32"},{"key":"305_CR36","doi-asserted-by":"crossref","unstructured":"Yeh Y T, Yang L, Watson M et al. Synthesizing open worlds with constraints using locally annealed reversible jump MCMC. ACM Transactions on Graphics, 2012, 31(4): Article No. 56.","DOI":"10.1145\/2185520.2185552"},{"issue":"3","key":"305_CR37","doi-asserted-by":"publisher","first-page":"313","DOI":"10.1145\/882262.882269","volume":"22","author":"P P\u00e9rez","year":"2003","unstructured":"P\u00e9rez P, Gangnet M, Blake A. Poisson image editing. ACM Transactions on Graphics, 2003, 22(3): 313-318.","journal-title":"ACM Transactions on Graphics"},{"key":"305_CR38","doi-asserted-by":"crossref","unstructured":"Liao Z, Karsch K, Forsyth D. An approximate shading model for object relighting. In Proc. the IEEE Conference on Computer Vission and Pattern Recognition, June 2015, pp.5307-5314.","DOI":"10.1109\/CVPR.2015.7299168"},{"issue":"1","key":"305_CR39","doi-asserted-by":"publisher","first-page":"423","DOI":"10.1146\/annurev-vision-091517-034110","volume":"4","author":"JH Elder","year":"2018","unstructured":"Elder J H. Shape from contour: Computation and representation. Annual Review of Vision Science, 2018, 4(1): 423-450.","journal-title":"Annual Review of Vision Science"},{"key":"305_CR40","doi-asserted-by":"crossref","unstructured":"Johnston S F. Lumo: Illumination for cel animation. In Proc. the 2nd International Symposium on Non-Photorealistic Animation and Rendering, June 2002, pp.45-52.","DOI":"10.1145\/508530.508538"},{"key":"305_CR41","doi-asserted-by":"crossref","unstructured":"Wu T P, Sun J, Tang C K et al. Interactive normal reconstruction from a single image. ACM Transactions on Graphics, 2008, 27(5): Article No. 119.","DOI":"10.1145\/1409060.1409072"},{"key":"305_CR42","doi-asserted-by":"crossref","unstructured":"Grosse R, Johnson M K, Adelson E H et al. Ground truth dataset and baseline evaluations for intrinsic image algorithms. In Proc. the 12th IEEE International Conference on Computer Vision, September 2009, pp.2335-2342.","DOI":"10.1109\/ICCV.2009.5459428"},{"key":"305_CR43","doi-asserted-by":"crossref","unstructured":"Karsch K, Sunkavalli K, Hadap S et al. Automatic scene inference for 3D object compositing. ACM Transactions on Graphics, 2014, 33(3): Article No. 32.","DOI":"10.1145\/2602146"},{"key":"305_CR44","doi-asserted-by":"crossref","unstructured":"Godard C, Aodha M O, Brostow G J. Unsupervised monocular depth estimation with left-right consistency. In Proc. the IEEE Conference on Computer Vision and Pattern Recognition, July 2017, pp.6602-6611.","DOI":"10.1109\/CVPR.2017.699"}],"container-title":["Journal of Computer Science and Technology"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11390-020-0305-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11390-020-0305-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11390-020-0305-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,6,8]],"date-time":"2020-06-08T21:06:05Z","timestamp":1591650365000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11390-020-0305-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,5]]},"references-count":44,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2020,5]]}},"alternative-id":["305"],"URL":"https:\/\/doi.org\/10.1007\/s11390-020-0305-9","relation":{},"ISSN":["1000-9000","1860-4749"],"issn-type":[{"value":"1000-9000","type":"print"},{"value":"1860-4749","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020,5]]},"assertion":[{"value":"15 January 2020","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 April 2020","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 May 2020","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}