{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,12]],"date-time":"2026-06-12T21:56:56Z","timestamp":1781301416940,"version":"3.54.1"},"reference-count":36,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.patcog.2026.113674","type":"journal-article","created":{"date-parts":[[2026,4,3]],"date-time":"2026-04-03T01:45:41Z","timestamp":1775180741000},"page":"113674","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PB","title":["SDGScenes: User-intent driven indoor scene generation via semantic dependency graph"],"prefix":"10.1016","volume":"179","author":[{"given":"Yuhan","family":"Gao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xueying","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jicong","family":"Ao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chao","family":"Yu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peng","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8379-9385","authenticated-orcid":false,"given":"Chenjia","family":"Bai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"3","key":"10.1016\/j.patcog.2026.113674_b1","doi-asserted-by":"crossref","first-page":"594","DOI":"10.1007\/s11390-019-1929-5","article-title":"A survey of 3D indoor scene synthesis","volume":"34","author":"Zhang","year":"2019","journal-title":"J. Comput. Sci. Tech."},{"key":"10.1016\/j.patcog.2026.113674_b2","series-title":"3D scene generation: A survey","author":"Wen","year":"2025"},{"key":"10.1016\/j.patcog.2026.113674_b3","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2025.103624","article-title":"Embodied intelligence for 3D understanding: A survey on 3D scene question answering","volume":"126","author":"Li","year":"2026","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.patcog.2026.113674_b4","series-title":"Advances in Neural Information Processing Systems","first-page":"12013","article-title":"ATISS: Autoregressive transformers for indoor scene synthesis","volume":"34","author":"Paschalidou","year":"2021"},{"key":"10.1016\/j.patcog.2026.113674_b5","doi-asserted-by":"crossref","unstructured":"J. Tang, Y. Nie, L. Markhasin, A. Dai, J. Thies, M. Nie\u00dfner, DiffuScene: Denoising Diffusion Models for Generative Indoor Scene Synthesis, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2024, pp. 20507\u201320518.","DOI":"10.1109\/CVPR52733.2024.01938"},{"key":"10.1016\/j.patcog.2026.113674_b6","doi-asserted-by":"crossref","unstructured":"H. Fu, B. Cai, L. Gao, L.-X. Zhang, J. Wang, C. Li, Q. Zeng, C. Sun, R. Jia, B. Zhao, H. Zhang, 3D-FRONT: 3D Furnished Rooms With layOuts and semaNTics, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, ICCV, 2021, pp. 10933\u201310942.","DOI":"10.1109\/ICCV48922.2021.01075"},{"key":"10.1016\/j.patcog.2026.113674_b7","first-page":"18225","article-title":"Layoutgpt: Compositional visual planning and generation with large language models","volume":"vol. 36","author":"Feng","year":"2023"},{"key":"10.1016\/j.patcog.2026.113674_b8","doi-asserted-by":"crossref","unstructured":"Y. Yang, F.-Y. Sun, L. Weihs, E. VanderBilt, A. Herrasti, W. Han, J. Wu, N. Haber, R. Krishna, L. Liu, C. Callison-Burch, M. Yatskar, A. Kembhavi, C. Clark, Holodeck: Language Guided Generation of 3D Embodied AI Environments, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2024, pp. 16227\u201316237.","DOI":"10.1109\/CVPR52733.2024.01536"},{"key":"10.1016\/j.patcog.2026.113674_b9","doi-asserted-by":"crossref","unstructured":"F.-Y. Sun, W. Liu, S. Gu, D. Lim, G. Bhat, F. Tombari, M. Li, N. Haber, J. Wu, LayoutVLM: Differentiable Optimization of 3D Layout via Vision-Language Models, in: Proceedings of the Computer Vision and Pattern Recognition Conference, CVPR, 2025, pp. 29469\u201329478.","DOI":"10.1109\/CVPR52734.2025.02744"},{"key":"10.1016\/j.patcog.2026.113674_b10","series-title":"Computer Vision \u2013 ECCV 2024","first-page":"52","article-title":"AnyHome: Open-vocabulary generation of structured and textured 3D homes","author":"Fu","year":"2025"},{"key":"10.1016\/j.patcog.2026.113674_b11","series-title":"Computer Vision \u2013 ECCV 2024 Workshops","first-page":"217","article-title":"I-Design: Personalized LLM interior designer","author":"\u00c7elen","year":"2025"},{"issue":"2","key":"10.1016\/j.patcog.2026.113674_b12","doi-asserted-by":"crossref","first-page":"2243","DOI":"10.1109\/TVCG.2025.3626731","article-title":"Chat2Layout: Interactive 3D furniture layout with a multimodal LLM","volume":"32","author":"Wang","year":"2026","journal-title":"IEEE Trans. Vis. Comput. Graphics"},{"key":"10.1016\/j.patcog.2026.113674_b13","series-title":"RoomCraft: Controllable and complete 3D indoor scene generation","author":"Zhou","year":"2025"},{"key":"10.1016\/j.patcog.2026.113674_b14","doi-asserted-by":"crossref","unstructured":"Y. Yang, B. Jia, P. Zhi, S. Huang, PhyScene: Physically Interactable 3D Scene Synthesis for Embodied AI, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2024, pp. 16262\u201316272.","DOI":"10.1109\/CVPR52733.2024.01539"},{"key":"10.1016\/j.patcog.2026.113674_b15","unstructured":"C. Lin, Y. M.U., InstructScene: Instruction-Driven 3D Indoor Scene Synthesis with Semantic Graph Prior, in: The Twelfth International Conference on Learning Representations, 2024."},{"key":"10.1016\/j.patcog.2026.113674_b16","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110562","article-title":"Granular3D: Delving into multi-granularity 3D scene graph prediction","volume":"153","author":"Huang","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113674_b17","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.111992","article-title":"SceneLLM: Implicit language reasoning in LLM for dynamic scene graph generation","volume":"170","author":"Zhang","year":"2026","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113674_b18","doi-asserted-by":"crossref","unstructured":"M. Deitke, D. Schwenk, J. Salvador, L. Weihs, O. Michel, E. VanderBilt, L. Schmidt, K. Ehsani, A. Kembhavi, A. Farhadi, Objaverse: A Universe of Annotated 3D Objects, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2023, pp. 13142\u201313153.","DOI":"10.1109\/CVPR52729.2023.01263"},{"key":"10.1016\/j.patcog.2026.113674_b19","first-page":"35799","article-title":"Objaverse-XL: A universe of 10m+ 3D objects","volume":"vol. 36","author":"Deitke","year":"2023"},{"key":"10.1016\/j.patcog.2026.113674_b20","doi-asserted-by":"crossref","unstructured":"H.I.D. Pun, H.I.I. Tam, A.T. Wang, X. Huo, A.X. Chang, M. Savva, HSM: Hierarchical Scene Motifs for Multi-Scale Indoor Scene Generation, in: Proceedings of the IEEE Conference on 3D Vision (3DV), 2026.","DOI":"10.1109\/3DV69130.2026.00131"},{"key":"10.1016\/j.patcog.2026.113674_b21","series-title":"Scenethesis: A language and vision agentic framework for 3D scene generation","author":"Ling","year":"2025"},{"key":"10.1016\/j.patcog.2026.113674_b22","doi-asserted-by":"crossref","unstructured":"H. Liu, C. Li, Y. Li, Y.J. Lee, Improved Baselines with Visual Instruction Tuning, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2024, pp. 26296\u201326306.","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"10.1016\/j.patcog.2026.113674_b23","first-page":"49250","article-title":"Instructblip: Towards general-purpose vision-language models with instruction tuning","volume":"vol. 36","author":"Dai","year":"2023"},{"key":"10.1016\/j.patcog.2026.113674_b24","article-title":"Mulberry: Empowering MLLM with o1-like reasoning and reflection via collective Monte Carlo tree search","volume":"vol. 38","author":"Yao","year":"2025"},{"key":"10.1016\/j.patcog.2026.113674_b25","unstructured":"Z. Zhao, H. Dong, A. Saha, C. Xiong, D. Sahoo, Automatic Curriculum Expert Iteration for Reliable LLM Reasoning, in: The Thirteenth International Conference on Learning Representations, 2025."},{"key":"10.1016\/j.patcog.2026.113674_b26","series-title":"Proceedings of the 36th International Conference on Neural Information Processing Systems","article-title":"Chain-of-thought prompting elicits reasoning in large language models","author":"Wei","year":"2022"},{"key":"10.1016\/j.patcog.2026.113674_b27","series-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","article-title":"Tree of thoughts: deliberate problem solving with large language models","author":"Yao","year":"2023"},{"key":"10.1016\/j.patcog.2026.113674_b28","unstructured":"M. Ni, Y. Fan, L. Zhang, W. Zuo, Visual-O1: Understanding Ambiguous Instructions via Multi-modal Multi-turn Chain-of-thoughts Reasoning, in: The Thirteenth International Conference on Learning Representations, 2025."},{"key":"10.1016\/j.patcog.2026.113674_b29","unstructured":"S. Ghosh, C.K.R. Evuru, S. Kumar, U. Tyagi, O. Nieto, Z. Jin, D. Manocha, Visual Description Grounding Reduces Hallucinations and Boosts Reasoning in LVLMs, in: The Thirteenth International Conference on Learning Representations, 2025."},{"key":"10.1016\/j.patcog.2026.113674_b30","unstructured":"Y. Hao, Y. Zhang, C. Fan, Planning Anything with Rigor: General-Purpose Zero-Shot Planning with LLM-based Formalized Programming, in: The Thirteenth International Conference on Learning Representations, 2025."},{"key":"10.1016\/j.patcog.2026.113674_b31","series-title":"Hunyuan3D 2.1: From images to high-fidelity 3D assets with production-ready PBR material","author":"Team","year":"2025"},{"key":"10.1016\/j.patcog.2026.113674_b32","series-title":"Proceedings of the 38th International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume":"139","author":"Radford","year":"2021"},{"key":"10.1016\/j.patcog.2026.113674_b33","series-title":"Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)","first-page":"3982","article-title":"Sentence-BERT: Sentence embeddings using siamese BERT-networks","author":"Reimers","year":"2019"},{"key":"10.1016\/j.patcog.2026.113674_b34","doi-asserted-by":"crossref","unstructured":"W. Deng, M. Qi, H. Ma, Global-Local Tree Search in VLMs for 3D Indoor Scene Generation, in: Proceedings of the Computer Vision and Pattern Recognition Conference, CVPR, 2025, pp. 8975\u20138984.","DOI":"10.1109\/CVPR52734.2025.00839"},{"key":"10.1016\/j.patcog.2026.113674_b35","doi-asserted-by":"crossref","first-page":"13842","DOI":"10.1109\/TASE.2025.3555559","article-title":"Sim2Real learning with domain randomization for autonomous guidewire navigation in robotic-assisted endovascular procedures","volume":"22","author":"Yao","year":"2025","journal-title":"IEEE Trans. Autom. Sci. Eng."},{"key":"10.1016\/j.patcog.2026.113674_b36","series-title":"RoboTwin 2.0: A scalable data generator and benchmark with strong domain randomization for robust bimanual robotic manipulation","author":"Chen","year":"2025"}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326006394?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326006394?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,12]],"date-time":"2026-06-12T21:16:05Z","timestamp":1781298965000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326006394"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":36,"alternative-id":["S0031320326006394"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113674","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"SDGScenes: User-intent driven indoor scene generation via semantic dependency graph","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113674","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"113674"}}