{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,21]],"date-time":"2026-02-21T19:17:33Z","timestamp":1771701453468,"version":"3.50.1"},"publisher-location":"Singapore","reference-count":41,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819785070","type":"print"},{"value":"9789819785087","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,3]],"date-time":"2024-11-03T00:00:00Z","timestamp":1730592000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,3]],"date-time":"2024-11-03T00:00:00Z","timestamp":1730592000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-97-8508-7_1","type":"book-chapter","created":{"date-parts":[[2024,11,2]],"date-time":"2024-11-02T06:02:22Z","timestamp":1730527342000},"page":"3-17","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Visual Harmony: LLM\u2019s Power in\u00a0Crafting Coherent Indoor Scenes from\u00a0Images"],"prefix":"10.1007","author":[{"given":"Genghao","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuxi","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chuanchen","family":"Luo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shibiao","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yue","family":"Ming","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junran","family":"Peng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Man","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,3]]},"reference":[{"key":"1_CR1","doi-asserted-by":"crossref","unstructured":"Alkhatib, Y.J., Forte, A., Bitelli, G., Pierdicca, R., Malinverni, E.: Bringing back lost heritage into life by 3d reconstruction in metaverse and virtual environments: the case study of Palmyra, Syria. In: International Conference on Extended Reality, pp. 91\u2013106 (2023)","DOI":"10.1007\/978-3-031-43404-4_7"},{"key":"1_CR2","unstructured":"Bhat, S.F., Birkl, R., Wofk, D., Wonka, P., M\u00fcller, M.: Zoedepth: zero-shot transfer by combining relative and metric depth (2023). arXiv:2302.12288"},{"key":"1_CR3","doi-asserted-by":"publisher","first-page":"42","DOI":"10.1016\/j.culher.2009.02.006","volume":"11","author":"F Bruno","year":"2010","unstructured":"Bruno, F., Bruno, S., De Sensi, G., Luchi, M.L., Mancuso, S., Muzzupappa, M.: From 3d reconstruction to virtual reality: a complete methodology for digital archaeological exhibition. J. Cult. Herit. 11, 42\u201349 (2010)","journal-title":"J. Cult. Herit."},{"key":"1_CR4","doi-asserted-by":"crossref","unstructured":"Chang, A., Monroe, W., Savva, M., Potts, C., Manning, C.D.: Text to 3d scene generation with rich lexical grounding. In: Proceedings of the 53rd Annual Meeting of the Association for Computational Linguistics and the 7th International Joint Conference on Natural Language Processing, pp. 53\u201362 (2015)","DOI":"10.3115\/v1\/P15-1006"},{"key":"1_CR5","doi-asserted-by":"crossref","unstructured":"Chen, S., Zhang, K., Shi, Y., Wang, H., Zhu, Y., Song, G., An, S., Kristjansson, J., Yang, X., Zwicker, M.: Panic-3d: stylized single-view 3d reconstruction from portraits of anime characters. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 21068\u201321077 (2023)","DOI":"10.1109\/CVPR52729.2023.02018"},{"key":"1_CR6","first-page":"17864","volume":"34","author":"B Cheng","year":"2021","unstructured":"Cheng, B., Schwing, A., Kirillov, A.: Per-pixel classification is not all you need for semantic segmentation. Adv. Neural. Inf. Process. Syst. 34, 17864\u201317875 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1_CR7","doi-asserted-by":"crossref","unstructured":"Di, X., Yu, P., Zhu, H., Cai, L., Sheng, Q., Sun, C., Ran, L.: Structural plan of indoor scenes with personalized preferences. In: Proceedings of the European Conference on Computer Vision, pp. 455\u2013468 (2020)","DOI":"10.1007\/978-3-030-66823-5_27"},{"key":"1_CR8","unstructured":"Feng, W., Zhu, W., Fu, T.J., Jampani, V., Akula, A., He, X., Basu, S., Wang, X.E., Wang, W.Y.: Layoutgpt: compositional visual planning and generation with large language models. In: Advances in Neural Information Processing Systems, pp. 18225\u201318250 (2023)"},{"key":"1_CR9","doi-asserted-by":"publisher","first-page":"129","DOI":"10.1016\/j.culher.2019.12.004","volume":"43","author":"D Ferdani","year":"2020","unstructured":"Ferdani, D., Fanini, B., Piccioli, M.C., Carboni, F., Vigliarolo, P.: 3d reconstruction and validation of historical background for immersive VR applications and games: the case study of the forum of augustus in rome. J. Cult. Herit. 43, 129\u2013143 (2020)","journal-title":"J. Cult. Herit."},{"key":"1_CR10","doi-asserted-by":"crossref","unstructured":"Gao, D., Rozenberszki, D., Leutenegger, S., Dai, A.: Diffcad: weakly-supervised probabilistic cad model retrieval and alignment from an RGB image (2023). arXiv:2311.18610","DOI":"10.1145\/3658236"},{"key":"1_CR11","doi-asserted-by":"crossref","unstructured":"G\u00fcmeli, C., Dai, A., Nie\u00dfner, M.: Roca: robust cad model retrieval and alignment from a single image. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4022\u20134031 (2022)","DOI":"10.1109\/CVPR52688.2022.00399"},{"key":"1_CR12","doi-asserted-by":"publisher","DOI":"10.1016\/j.optcom.2022.128894","volume":"526","author":"L He","year":"2023","unstructured":"He, L., Liu, K., He, Z., Cao, L.: Three-dimensional holographic communication system for the metaverse. Opt. Commun. 526, 128894 (2023)","journal-title":"Opt. Commun."},{"key":"1_CR13","doi-asserted-by":"publisher","first-page":"301","DOI":"10.1016\/j.isprsjprs.2022.02.014","volume":"186","author":"L Huan","year":"2022","unstructured":"Huan, L., Zheng, X., Gong, J.: Georec: geometry-enhanced semantic 3d reconstruction of RGB-d indoor scenes. ISPRS J. Photogramm. Remote. Sens. 186, 301\u2013314 (2022)","journal-title":"ISPRS J. Photogramm. Remote. Sens."},{"key":"1_CR14","unstructured":"Huang, S., Qi, S., Xiao, Y., Zhu, Y., Wu, Y.N., Zhu, S.C.: Cooperative holistic scene understanding: unifying 3d object, layout, and camera pose estimation. In: Proceedings of the 32nd International Conference on Neural Information Processing Systems, pp. 206\u2013217 (2018)"},{"key":"1_CR15","doi-asserted-by":"crossref","unstructured":"Huang, S., Qi, S., Zhu, Y., Xiao, Y., Xu, Y., Zhu, S.C.: Holistic 3d scene parsing and reconstruction from a single RGB image. In: Proceedings of the European Conference on Computer Vision, pp. 187\u2013203 (2018)","DOI":"10.1007\/978-3-030-01234-2_12"},{"key":"1_CR16","doi-asserted-by":"crossref","unstructured":"Izadinia, H., Shan, Q., Seitz, S.M.: Im2cad. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5134\u20135143 (2017)","DOI":"10.1109\/CVPR.2017.260"},{"key":"1_CR17","doi-asserted-by":"crossref","unstructured":"Jin, L., Zhang, J., Hold-Geoffroy, Y., Wang, O., Blackburn-Matzen, K., Sticha, M., Fouhey, D.F.: Perspective fields for single image camera calibration. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17307\u201317316 (2023)","DOI":"10.1109\/CVPR52729.2023.01660"},{"key":"1_CR18","doi-asserted-by":"crossref","unstructured":"Kumar, H., Khargonkar, N., Prabhakaran, B.: Cis2vr: CNN-based indoor scan to VR environment authoring framework. In: IEEE International Conference on Artificial Intelligence and Extended and Virtual Reality, pp. 128\u2013137 (2024)","DOI":"10.1109\/AIxVR59861.2024.00025"},{"key":"1_CR19","doi-asserted-by":"crossref","unstructured":"Kuo, W., Angelova, A., Lin, T.Y., Dai, A.: Mask2cad: 3d shape prediction by learning to segment and retrieve. In: Proceedings of the European Conference on Computer Vision, pp. 260\u2013277 (2020)","DOI":"10.1007\/978-3-030-58580-8_16"},{"key":"1_CR20","unstructured":"Langer, F., Budvytis, I., Cipolla, R.: Sparse multi-object render-and-compare (2023). arXiv:2310.11184"},{"key":"1_CR21","first-page":"1","volume":"38","author":"M Li","year":"2019","unstructured":"Li, M., Patil, A.G., Xu, K., Chaudhuri, S., Khan, O., Shamir, A., Tu, C., Chen, B., Cohen-Or, D., Zhang, H.: Grains: generative recursive autoencoders for indoor scenes. ACM Trans. Graph. 38, 1\u201316 (2019)","journal-title":"ACM Trans. Graph."},{"key":"1_CR22","doi-asserted-by":"crossref","unstructured":"Liu, H., Zheng, Y., Chen, G., Cui, S., Han, X.: Towards high-fidelity single-view holistic reconstruction of indoor scenes. In: Proceedings of the European Conference on Computer Vision, pp. 429\u2013446 (2022)","DOI":"10.1007\/978-3-031-19769-7_25"},{"key":"1_CR23","doi-asserted-by":"crossref","unstructured":"Luo, A., Zhang, Z., Wu, J., Tenenbaum, J.B.: End-to-end optimization of scene layout. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3754\u20133763 (2020)","DOI":"10.1109\/CVPR42600.2020.00381"},{"key":"1_CR24","doi-asserted-by":"publisher","first-page":"116","DOI":"10.1016\/j.cag.2021.07.014","volume":"100","author":"A Manni","year":"2021","unstructured":"Manni, A., Oriti, D., Sanna, A., De Pace, F., Manuri, F.: Snap2cad: 3d indoor environment reconstruction for AR\/VR applications using a smartphone device. Comput. Graph. 100, 116\u2013124 (2021)","journal-title":"Comput. Graph."},{"key":"1_CR25","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2020.107271","volume":"103","author":"Y Nie","year":"2020","unstructured":"Nie, Y., Guo, S., Chang, J., Han, X., Huang, J., Hu, S.M., Zhang, J.J.: Shallow2deep: indoor scene modeling by single image understanding. Pattern Recogn. 103, 107271 (2020)","journal-title":"Pattern Recogn."},{"key":"1_CR26","doi-asserted-by":"crossref","unstructured":"Nie, Y., Han, X., Guo, S., Zheng, Y., Chang, J., Zhang, J.J.: Total 3d understanding: joint layout, object pose and mesh reconstruction for indoor scenes from a single image. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 55\u201364 (2020)","DOI":"10.1109\/CVPR42600.2020.00013"},{"key":"1_CR27","doi-asserted-by":"crossref","unstructured":"Purkait, P., Zach, C., Reid, I.: Sg-vae: scene grammar variational autoencoder to generate new indoor scenes. In: Proceedings of the European Conference on Computer Vision, pp. 155\u2013171 (2020)","DOI":"10.1007\/978-3-030-58586-0_10"},{"key":"1_CR28","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763 (2021)"},{"key":"1_CR29","doi-asserted-by":"crossref","unstructured":"Silberman, N., Hoiem, D., Kohli, P., Fergus, R.: Indoor segmentation and support inference from RGBD images. In: Proceedings of the European Conference on Computer Vision, pp. 746\u2013760 (2012)","DOI":"10.1007\/978-3-642-33715-4_54"},{"key":"1_CR30","doi-asserted-by":"crossref","unstructured":"Song, S., Lichtenberg, S.P., Xiao, J.: Sun RGB-d: a RGB-d scene understanding benchmark suite. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 567\u2013576 (2015)","DOI":"10.1109\/CVPR.2015.7298655"},{"key":"1_CR31","doi-asserted-by":"crossref","unstructured":"Sturm, J., Engelhard, N., Endres, F., Burgard, W., Cremers, D.: A benchmark for the evaluation of RGB-d slam systems. In: IEEE\/RSJ International Conference on Intelligent Robots and Systems, pp. 573\u2013580 (2012)","DOI":"10.1109\/IROS.2012.6385773"},{"issue":"1","key":"1_CR32","doi-asserted-by":"publisher","first-page":"9","DOI":"10.1007\/s44267-024-00042-1","volume":"2","author":"JM Sun","year":"2024","unstructured":"Sun, J.M., Wu, T., Gao, L.: Recent advances in implicit representation-based 3d shape generation. Visual Intell. 2(1), 9 (2024)","journal-title":"Visual Intell."},{"issue":"1","key":"1_CR33","doi-asserted-by":"publisher","first-page":"14","DOI":"10.1007\/s44267-024-00046-x","volume":"2","author":"Y Sun","year":"2024","unstructured":"Sun, Y., Zhang, X., Miao, Y.: A review of point cloud segmentation for understanding 3d indoor scenes. Visual Intell. 2(1), 14 (2024)","journal-title":"Visual Intell."},{"key":"1_CR34","first-page":"1","volume":"38","author":"K Wang","year":"2019","unstructured":"Wang, K., Lin, Y.A., Weissmann, B., Savva, M., Chang, A.X., Ritchie, D.: Planit: planning and instantiating indoor scenes with relation graph and spatial prior networks. ACM Trans. Graph. 38, 1\u201315 (2019)","journal-title":"ACM Trans. Graph."},{"key":"1_CR35","doi-asserted-by":"publisher","first-page":"9565","DOI":"10.1007\/s11042-019-08034-w","volume":"79","author":"X Xiao-lu","year":"2020","unstructured":"Xiao-lu, X.: Three-dimensional reconstruction based on multi-view photometric stereo fusion technology in movies special-effect. Multimedia Tools Appl. 79, 9565\u20139578 (2020)","journal-title":"Multimedia Tools Appl."},{"key":"1_CR36","doi-asserted-by":"crossref","unstructured":"Yan, K., Luan, F., Ha\u0161an, M., Groueix, T., Deschaintre, V., Zhao, S.: Psdr-room: single photo to scene using differentiable rendering. In: SIGGRAPH Asia, pp. 1\u201311 (2023)","DOI":"10.1145\/3610548.3618165"},{"key":"1_CR37","doi-asserted-by":"crossref","unstructured":"Yang, Y., Sun, F.Y., Weihs, L., VanderBilt, E., Herrasti, A., Han, W., Wu, J., Haber, N., Krishna, R., Liu, L., et\u00a0al.: Holodeck: language guided generation of 3d embodied AI environments. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 20\u201325 (2024)","DOI":"10.1109\/CVPR52733.2024.01536"},{"key":"1_CR38","doi-asserted-by":"crossref","unstructured":"Ye, E., Wang, Y., Zhang, H., Gao, Y., Wang, H., Sun, H.: Recovering a molecule\u2019s 3d dynamics from liquid-phase electron microscopy movies. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10767\u201310777 (2023)","DOI":"10.1109\/ICCV51070.2023.00988"},{"key":"1_CR39","doi-asserted-by":"crossref","unstructured":"Zhang, C., Cui, Z., Zhang, Y., Zeng, B., Pollefeys, M., Liu, S.: Holistic 3d scene understanding from a single image with implicit representation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8833\u20138842 (2021)","DOI":"10.1109\/CVPR46437.2021.00872"},{"key":"1_CR40","first-page":"1","volume":"22","author":"D Zhang","year":"2021","unstructured":"Zhang, D., Xu, F., Pun, C.M., Yang, Y., Lan, R., Wang, L., Li, Y., Gao, H.: Virtual reality aided high-quality 3d reconstruction by remote drones. ACM Trans. Internet Technol. 22, 1\u201320 (2021)","journal-title":"ACM Trans. Internet Technol."},{"key":"1_CR41","first-page":"1","volume":"39","author":"Z Zhang","year":"2020","unstructured":"Zhang, Z., Yang, Z., Ma, C., Luo, L., Huth, A., Vouga, E., Huang, Q.: Deep generative modeling for scene synthesis via hybrid representations. ACM Trans. Graph. 39, 1\u201321 (2020)","journal-title":"ACM Trans. Graph."}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition and Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-97-8508-7_1","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,2]],"date-time":"2024-11-02T06:11:58Z","timestamp":1730527918000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-97-8508-7_1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,3]]},"ISBN":["9789819785070","9789819785087"],"references-count":41,"URL":"https:\/\/doi.org\/10.1007\/978-981-97-8508-7_1","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,3]]},"assertion":[{"value":"3 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"PRCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Chinese Conference on Pattern Recognition and Computer Vision  (PRCV)","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Urumqi","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18 October 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"7","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ccprcv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/2024.prcv.cn\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}