{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T12:25:06Z","timestamp":1783081506235,"version":"3.54.6"},"reference-count":43,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100002701","name":"Ministry of Education","doi-asserted-by":"publisher","award":["250405448235212"],"award-info":[{"award-number":["250405448235212"]}],"id":[{"id":"10.13039\/501100002701","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002701","name":"Ministry of Education","doi-asserted-by":"publisher","award":["250405448255034"],"award-info":[{"award-number":["250405448255034"]}],"id":[{"id":"10.13039\/501100002701","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Journal of Visual Communication and Image Representation"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1016\/j.jvcir.2026.104814","type":"journal-article","created":{"date-parts":[[2026,5,9]],"date-time":"2026-05-09T00:03:22Z","timestamp":1778285002000},"page":"104814","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Graph visual representation for controllable scene layout generation"],"prefix":"10.1016","volume":"118","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0260-3169","authenticated-orcid":false,"given":"Jin","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-4251-9062","authenticated-orcid":false,"given":"Minghan","family":"Ma","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0720-2505","authenticated-orcid":false,"given":"Longjiang","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-2859-3376","authenticated-orcid":false,"given":"Xue","family":"Dong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Meirui","family":"Ren","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.jvcir.2026.104814_b1","series-title":"European Conference on Computer Vision","first-page":"93","article-title":"Latent guard: a safety framework for text-to-image generation","author":"Liu","year":"2025"},{"key":"10.1016\/j.jvcir.2026.104814_b2","series-title":"European Conference on Computer Vision","first-page":"432","article-title":"Be yourself: Bounded attention for multi-subject text-to-image generation","author":"Dahary","year":"2025"},{"key":"10.1016\/j.jvcir.2026.104814_b3","article-title":"Raphael: Text-to-image generation via large mixture of diffusion paths","volume":"36","author":"Xue","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.jvcir.2026.104814_b4","article-title":"Subject-driven text-to-image generation via apprenticeship learning","volume":"36","author":"Chen","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.jvcir.2026.104814_b5","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1016\/j.jvcir.2017.07.001","article-title":"Learning location constrained pixel classifiers for image parsing","volume":"49","author":"Dang","year":"2017","journal-title":"J. Vis. Commun. Image Represent."},{"key":"10.1016\/j.jvcir.2026.104814_b6","series-title":"Learning to navigate in complex environments","author":"Mirowski","year":"2016"},{"key":"10.1016\/j.jvcir.2026.104814_b7","doi-asserted-by":"crossref","unstructured":"C. Jia, M. Luo, Z. Dang, G. Dai, X. Chang, M. Wang, J. Wang, Ssmg: Spatial-semantic map guided diffusion model for free-form layout-to-image generation, in: Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 38, 2024, pp. 2480\u20132488.","DOI":"10.1609\/aaai.v38i3.28024"},{"key":"10.1016\/j.jvcir.2026.104814_b8","doi-asserted-by":"crossref","unstructured":"J. Cho, L. Li, Z. Yang, Z. Gan, L. Wang, M. Bansal, Diagnostic benchmark and iterative inpainting for layout-guided image generation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 5280\u20135289.","DOI":"10.1109\/CVPRW63382.2024.00537"},{"key":"10.1016\/j.jvcir.2026.104814_b9","doi-asserted-by":"crossref","first-page":"86092","DOI":"10.1109\/ACCESS.2022.3198686","article-title":"A conditional deep framework for automatic layout generation","volume":"10","author":"Shi","year":"2022","journal-title":"IEEE Access"},{"key":"10.1016\/j.jvcir.2026.104814_b10","doi-asserted-by":"crossref","DOI":"10.1016\/j.jvcir.2023.103963","article-title":"Undirected graph representing strategy for general room layout estimation","volume":"97","author":"Yao","year":"2023","journal-title":"J. Vis. Commun. Image Represent."},{"key":"10.1016\/j.jvcir.2026.104814_b11","unstructured":"C.-F. Yang, W.-C. Fan, F.-E. Yang, Y.-C.F. Wang, Layouttransformer: Scene layout generation with conceptual and spatial diversity, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021, pp. 3732\u20133741."},{"key":"10.1016\/j.jvcir.2026.104814_b12","series-title":"Layoutgan: Generating graphic layouts with wireframe discriminators","author":"Li","year":"2019"},{"key":"10.1016\/j.jvcir.2026.104814_b13","series-title":"Composition-aware graphic layout GAN for visual-textual presentation designs","author":"Zhou","year":"2022"},{"key":"10.1016\/j.jvcir.2026.104814_b14","doi-asserted-by":"crossref","DOI":"10.1016\/j.compeleceng.2021.107269","article-title":"A study on the automatic generation of banner layouts","volume":"93","author":"Hu","year":"2021","journal-title":"Comput. Electr. Eng."},{"key":"10.1016\/j.jvcir.2026.104814_b15","article-title":"Visual programming for step-by-step text-to-image generation and evaluation","volume":"36","author":"Cho","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.jvcir.2026.104814_b16","doi-asserted-by":"crossref","unstructured":"N. Inoue, K. Kikuchi, E. Simo-Serra, M. Otani, K. Yamaguchi, Layoutdm: Discrete diffusion model for controllable layout generation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 10167\u201310176.","DOI":"10.1109\/CVPR52729.2023.00980"},{"key":"10.1016\/j.jvcir.2026.104814_b17","series-title":"European Conference on Computer Vision","first-page":"474","article-title":"Blt: Bidirectional layout transformer for controllable layout generation","author":"Kong","year":"2022"},{"key":"10.1016\/j.jvcir.2026.104814_b18","doi-asserted-by":"crossref","unstructured":"J. Schult, S. Tsai, L. H\u00f6llein, B. Wu, J. Wang, C.-Y. Ma, K. Li, X. Wang, F. Wimbauer, Z. He, et al., Controlroom3d: Room generation using semantic proxy rooms, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 6201\u20136210.","DOI":"10.1109\/CVPR52733.2024.00593"},{"key":"10.1016\/j.jvcir.2026.104814_b19","doi-asserted-by":"crossref","unstructured":"A.G. Patil, O. Ben-Eliezer, O. Perel, H. Averbuch-Elor, Read: Recursive autoencoders for document layout generation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops, 2020, pp. 544\u2013545.","DOI":"10.1109\/CVPRW50498.2020.00280"},{"key":"10.1016\/j.jvcir.2026.104814_b20","unstructured":"R. Socher, C.C. Lin, C. Manning, A.Y. Ng, Parsing natural scenes and natural language with recursive neural networks, in: Proceedings of the 28th International Conference on Machine Learning, ICML-11, 2011, pp. 129\u2013136."},{"key":"10.1016\/j.jvcir.2026.104814_b21","unstructured":"K. Yamaguchi, Canvasvae: Learning to generate vector graphic documents, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 5481\u20135489."},{"key":"10.1016\/j.jvcir.2026.104814_b22","doi-asserted-by":"crossref","unstructured":"Z. Jiang, S. Sun, J. Zhu, J.-G. Lou, D. Zhang, Coarse-to-fine generative modeling for graphic layouts, in: Proceedings of the AAAI Conference on Artificial Intelligence, 2022, pp. 1096\u20131103.","DOI":"10.1609\/aaai.v36i1.19994"},{"key":"10.1016\/j.jvcir.2026.104814_b23","doi-asserted-by":"crossref","unstructured":"K. Kikuchi, E. Simo-Serra, M. Otani, K. Yamaguchi, Constrained graphic layout generation via latent optimization, in: Proceedings of the 29th ACM International Conference on Multimedia, 2021, pp. 88\u201396.","DOI":"10.1145\/3474085.3475497"},{"key":"10.1016\/j.jvcir.2026.104814_b24","article-title":"Pastegan: A semi-parametric method to generate image from scene graph","volume":"32","author":"Li","year":"2019","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.jvcir.2026.104814_b25","series-title":"2022 IEEE International Conference on Multimedia and Expo","first-page":"1","article-title":"High-quality image generation from scene graphs with transformer","author":"Zhao","year":"2022"},{"key":"10.1016\/j.jvcir.2026.104814_b26","doi-asserted-by":"crossref","unstructured":"P. Esser, R. Rombach, B. Ommer, Taming transformers for high-resolution image synthesis, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021, pp. 12873\u201312883.","DOI":"10.1109\/CVPR46437.2021.01268"},{"key":"10.1016\/j.jvcir.2026.104814_b27","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2024.127730","article-title":"Element-conditioned GAN for graphic layout generation","volume":"591","author":"Chen","year":"2024","journal-title":"Neurocomputing"},{"key":"10.1016\/j.jvcir.2026.104814_b28","doi-asserted-by":"crossref","unstructured":"D.M. Arroyo, J. Postels, F. Tombari, Variational transformer networks for layout generation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021, pp. 13642\u201313652.","DOI":"10.1109\/CVPR46437.2021.01343"},{"key":"10.1016\/j.jvcir.2026.104814_b29","doi-asserted-by":"crossref","first-page":"477","DOI":"10.1016\/j.jvcir.2018.12.027","article-title":"Scene graph captioner: Image captioning based on structural visual representation","volume":"58","author":"Xu","year":"2019","journal-title":"J. Vis. Commun. Image Represent."},{"key":"10.1016\/j.jvcir.2026.104814_b30","doi-asserted-by":"crossref","unstructured":"J. Johnson, A. Gupta, L. Fei-Fei, Image generation from scene graphs, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018, pp. 1219\u20131228.","DOI":"10.1109\/CVPR.2018.00133"},{"key":"10.1016\/j.jvcir.2026.104814_b31","series-title":"Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XXVI 16","first-page":"210","article-title":"Learning canonical representations for scene graph to image generation","author":"Herzig","year":"2020"},{"key":"10.1016\/j.jvcir.2026.104814_b32","series-title":"Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part III 16","first-page":"491","article-title":"Neural design network: Graphic layout generation with constraints","author":"Lee","year":"2020"},{"key":"10.1016\/j.jvcir.2026.104814_b33","series-title":"International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.jvcir.2026.104814_b34","series-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2018"},{"key":"10.1016\/j.jvcir.2026.104814_b35","doi-asserted-by":"crossref","unstructured":"H. Caesar, J. Uijlings, V. Ferrari, Coco-stuff: Thing and stuff classes in context, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018, pp. 1209\u20131218.","DOI":"10.1109\/CVPR.2018.00132"},{"key":"10.1016\/j.jvcir.2026.104814_b36","doi-asserted-by":"crossref","unstructured":"Y. Li, W. Ouyang, B. Zhou, K. Wang, X. Wang, Scene graph generation from objects, phrases and region captions, in: Proceedings of the IEEE International Conference on Computer Vision, 2017, pp. 1261\u20131270.","DOI":"10.1109\/ICCV.2017.142"},{"key":"10.1016\/j.jvcir.2026.104814_b37","doi-asserted-by":"crossref","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","article-title":"Visual genome: Connecting language and vision using crowdsourced dense image annotations","volume":"123","author":"Krishna","year":"2017","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.jvcir.2026.104814_b38","article-title":"Layoutgpt: Compositional visual planning and generation with large language models","volume":"36","author":"Feng","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.jvcir.2026.104814_b39","doi-asserted-by":"crossref","DOI":"10.1109\/ACCESS.2024.3452957","article-title":"Text2Layout: Layout generation from text representation using transformer","volume":"12","author":"Takahashi","year":"2024","journal-title":"IEEE Access"},{"key":"10.1016\/j.jvcir.2026.104814_b40","series-title":"Software Engineering Perspectives in Intelligent Systems: Proceedings of 4th Computational Methods in Systems and Software 2020, Vol. 1 4","first-page":"102","article-title":"Quality assessment method for GAN based on modified metrics inception score and Fr\u00e9chet inception distance","author":"Obukhov","year":"2020"},{"key":"10.1016\/j.jvcir.2026.104814_b41","doi-asserted-by":"crossref","unstructured":"R. Zhang, P. Isola, A.A. Efros, E. Shechtman, O. Wang, The unreasonable effectiveness of deep features as a perceptual metric, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018, pp. 586\u2013595.","DOI":"10.1109\/CVPR.2018.00068"},{"key":"10.1016\/j.jvcir.2026.104814_b42","series-title":"One weird trick for parallelizing convolutional neural networks","author":"Krizhevsky","year":"2014"},{"key":"10.1016\/j.jvcir.2026.104814_b43","article-title":"GLIGEN: Open-set grounded text-to-image generation","author":"Li","year":"2023","journal-title":"CVPR"}],"container-title":["Journal of Visual Communication and Image Representation"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1047320326001094?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1047320326001094?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T12:05:30Z","timestamp":1783080330000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1047320326001094"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":43,"alternative-id":["S1047320326001094"],"URL":"https:\/\/doi.org\/10.1016\/j.jvcir.2026.104814","relation":{},"ISSN":["1047-3203"],"issn-type":[{"value":"1047-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,6]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Graph visual representation for controllable scene layout generation","name":"articletitle","label":"Article Title"},{"value":"Journal of Visual Communication and Image Representation","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.jvcir.2026.104814","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier Inc.","name":"copyright","label":"Copyright"}],"article-number":"104814"}}