{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,17]],"date-time":"2026-08-17T15:49:26Z","timestamp":1786981766240,"version":"build-2736575974"},"reference-count":66,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2026,2,21]],"date-time":"2026-02-21T00:00:00Z","timestamp":1771632000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,2,21]],"date-time":"2026-02-21T00:00:00Z","timestamp":1771632000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U21B2042"],"award-info":[{"award-number":["U21B2042"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62320106010"],"award-info":[{"award-number":["62320106010"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Foundation of Shunde Young Investigator Program","award":["2024A1515110065"],"award-info":[{"award-number":["2024A1515110065"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2026,3]]},"DOI":"10.1007\/s11263-025-02634-w","type":"journal-article","created":{"date-parts":[[2026,2,21]],"date-time":"2026-02-21T07:15:24Z","timestamp":1771658124000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["FurniScene: A Large-scale 3D Room Dataset with Intricate Furnishing Scenes"],"prefix":"10.1007","volume":"134","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1579-2357","authenticated-orcid":false,"given":"Yuxi","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Junran","family":"Peng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Genghao","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chuanchen","family":"Luo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shibiao","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Man","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhaoxiang","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,2,21]]},"reference":[{"key":"2634_CR1","doi-asserted-by":"crossref","unstructured":"Armeni, I., Sener, O., Zamir, A. R., Jiang, H., Brilakis, I., Fischer, M., & Savarese, S. (2016). 3d semantic parsing of large-scale indoor spaces. In Proc. CVPR, 1534\u20131543.","DOI":"10.1109\/CVPR.2016.170"},{"key":"2634_CR2","doi-asserted-by":"crossref","unstructured":"Avetisyan, A., Dahnert, M., Dai, A., Savva, M., Chang, A. X., & Nie\u00dfner, M. (2019). Scan2cad: Learning cad model alignment in rgb-d scans. In Proc. CVPR, 2614\u20132623.","DOI":"10.1109\/CVPR.2019.00272"},{"key":"2634_CR3","doi-asserted-by":"crossref","unstructured":"Chang, A., Dai, A., Funkhouser, T., Halber, M., Niebner, M., Savva, M., Song, S., Zeng, A., & Zhang, Y. (2017). Matterport3d: Learning from rgb-d data in indoor environments. In Proc. 3DV, 667\u2013676.","DOI":"10.1109\/3DV.2017.00081"},{"key":"2634_CR4","doi-asserted-by":"crossref","unstructured":"Dai, A., Chang, A. X., Savva, M., Halber, M., Funkhouser, T., & Nie\u00dfner, M. (2017). Scannet: Richly-annotated 3d reconstructions of indoor scenes. In Proc. CVPR, 5828\u20135839.","DOI":"10.1109\/CVPR.2017.261"},{"key":"2634_CR5","doi-asserted-by":"crossref","unstructured":"Delitzas, A., Takmaz, A., Tombari, F., Sumner, R., Pollefeys, M., & Engelmann, F. (2024). Scenefun3d: fine-grained functionality and affordance understanding in 3d scenes. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 14531\u201314542.","DOI":"10.1109\/CVPR52733.2024.01377"},{"issue":"4","key":"2634_CR6","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/2010324.1964929","volume":"30","author":"M Fisher","year":"2011","unstructured":"Fisher, M., Savva, M., & Hanrahan, P. (2011). Characterizing structural relationships in scenes using graph kernels. ACM Transactions on Graphics, 30(4), 1\u201312.","journal-title":"ACM Transactions on Graphics"},{"issue":"3","key":"2634_CR7","doi-asserted-by":"publisher","first-page":"323","DOI":"10.1109\/TG.2019.2957733","volume":"12","author":"J Freiknecht","year":"2019","unstructured":"Freiknecht, J., & Effelsberg, W. (2019). Procedural generation of multistory buildings with interior. IEEE Transactions on Games, 12(3), 323\u2013336.","journal-title":"IEEE Transactions on Games"},{"key":"2634_CR8","doi-asserted-by":"crossref","unstructured":"Fu, H., Cai, B., Gao, L., Zhang, L.-X., Wang, J., Li, C., Zeng, Q., Sun, C., Jia, R., Zhao, B., et al. (2021). 3d-front: 3d furnished rooms with layouts and semantics. In Proc. ICCV, 10933\u201310942.","DOI":"10.1109\/ICCV48922.2021.01075"},{"issue":"07","key":"2634_CR9","doi-asserted-by":"publisher","first-page":"8902","DOI":"10.1109\/TPAMI.2023.3237577","volume":"45","author":"L Gao","year":"2023","unstructured":"Gao, L., Sun, J.-M., Mo, K., Lai, Y.-K., Guibas, L. J., & Yang, J. (2023). Scenehgn: Hierarchical graph networks for 3d indoor scene generation with fine-grained geometry. IEEE Transactions on Pattern Analysis and Machine Intelligence, 45(07), 8902\u20138919.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2634_CR10","doi-asserted-by":"crossref","unstructured":"Handa, A., Patraucean, V., Badrinarayanan, V., Stent, S., & Cipolla, R. (2016). Understanding real world indoor scenes with synthetic data. In Proc. CVPR, 4077\u20134085.","DOI":"10.1109\/CVPR.2016.442"},{"key":"2634_CR11","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., & Abbeel, P. (2020). Denoising diffusion probabilistic models. In Proc. NeurIPS, 33, 6840\u20136851.","journal-title":"In Proc. NeurIPS"},{"key":"2634_CR12","doi-asserted-by":"crossref","unstructured":"Holmquist, K. & Wandt, B. (2023). Diffpose: Multi-hypothesis human pose estimation using diffusion models. In Proc. ICCV, 15977\u201315987.","DOI":"10.1109\/ICCV51070.2023.01464"},{"key":"2634_CR13","unstructured":"Hu, S., Arroyo, D. M., Debats, S., Manhardt, F., Carlone, L., & Tombari, F. (2024). Mixed diffusion for 3d indoor scene synthesis. arXiv preprint arXiv:2405.21066."},{"key":"2634_CR14","doi-asserted-by":"crossref","unstructured":"Hua, B.-S., Pham, Q.-H., Nguyen, D. T., Tran, M.-K., Yu, L.-F., & Yeung, S.-K. (2016). Scenenn: A scene meshes dataset with annotations. In Proc. 3DV, 92\u2013101.","DOI":"10.1109\/3DV.2016.18"},{"key":"2634_CR15","doi-asserted-by":"crossref","unstructured":"Huang, R., Lam, M., Wang, J., Su, D., Yu, D., Ren, Y., & Zhao, Z. (2022). Fastdiff: A fast conditional diffusion model for high-quality speech synthesis. In Proc. IJCAI, 4157\u20134163.","DOI":"10.24963\/ijcai.2022\/577"},{"key":"2634_CR16","doi-asserted-by":"crossref","unstructured":"Huang, Y., Yang, H., Luo, C., Wang, Y., Xu, S., Zhang, Z., Zhang, M., & Peng, J. (2024). Stablemofusion: Towards robust and efficient diffusion-based motion generation framework. In Proc. ACM MM, 224\u2013232.","DOI":"10.1145\/3664647.3681657"},{"key":"2634_CR17","doi-asserted-by":"crossref","unstructured":"Khanna, M., Mao, Y., Jiang, H., Haresh, S., Schacklett, B., Batra, D., Clegg, A., Undersander, E., Chang, A. X., & Savva, M. (2023). Habitat synthetic scenes dataset (hssd-200): An analysis of 3d scene scale and realism tradeoffs for objectgoal navigation. arXiv preprint arXiv:2306.11290.","DOI":"10.1109\/CVPR52733.2024.01550"},{"key":"2634_CR18","doi-asserted-by":"crossref","unstructured":"Kim, G., Kwon, T., & Ye, J. C. (2022). Diffusionclip: Text-guided diffusion models for robust image manipulation. In Proc. CVPR, 2426\u20132435.","DOI":"10.1109\/CVPR52688.2022.00246"},{"issue":"2","key":"2634_CR19","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3303766","volume":"38","author":"M Li","year":"2019","unstructured":"Li, M., Patil, A. G., Xu, K., Chaudhuri, S., Khan, O., Shamir, A., Tu, C., Chen, B., Cohen-Or, D., & Zhang, H. (2019). Grains: Generative recursive autoencoders for indoor scenes. ACM Transactions on Graphics, 38(2), 1\u201316.","journal-title":"ACM Transactions on Graphics"},{"key":"2634_CR20","doi-asserted-by":"crossref","unstructured":"Li, Z., Gan, R., Luo, C., Wang, Y., Liu, J., Zhu, Z., Li, Q., Yin, X., Zhang, M., Zhang, Z., et al. (2024). Materialseg3d: Segmenting dense materials from 2d priors for 3d assets. In Proc. ACM MM, 370\u2013379.","DOI":"10.1145\/3664647.3680757"},{"key":"2634_CR21","doi-asserted-by":"crossref","unstructured":"Li, Z., Yu, T.-W., Sang, S., Wang, S., Song, M., Liu, Y., Yeh, Y.-Y., Zhu, R., Gundavarapu, N., Shi, J., et al. (2021). Openrooms: An open framework for photorealistic indoor scene datasets. In Proc. CVPR, 7190\u20137199.","DOI":"10.1109\/CVPR46437.2021.00711"},{"key":"2634_CR22","doi-asserted-by":"publisher","first-page":"5173","DOI":"10.1609\/aaai.v39i5.32549","volume":"39","author":"Z Liang","year":"2025","unstructured":"Liang, Z., Xu, G., Wu, H., Huang, Y., Li, W., & Duan, L. (2025). S-inf: Towards realistic indoor scene synthesis via scene implicit neural field. In Proc. AAAI, 39, 5173\u20135181.","journal-title":"In Proc. AAAI"},{"key":"2634_CR23","doi-asserted-by":"crossref","unstructured":"Liao, J., Luo, C., Du, Y., Wang, Y., Yin, X., Zhang, M., Zhang, Z., & Peng, J. (2024). Hardmo: A large-scale hardcase dataset for motion capture. In Proc. CVPR, 1629\u20131638.","DOI":"10.1109\/CVPR52733.2024.00161"},{"key":"2634_CR24","volume-title":"Instructscene: Instruction-driven 3d indoor scene synthesis with semantic graph prior","author":"C Lin","year":"2023","unstructured":"Lin, C., & Yadong, M. (2023). Instructscene: Instruction-driven 3d indoor scene synthesis with semantic graph prior. ICLR: In Proc."},{"key":"2634_CR25","doi-asserted-by":"publisher","first-page":"11020","DOI":"10.1609\/aaai.v36i10.21350","volume":"36","author":"J Liu","year":"2022","unstructured":"Liu, J., Li, C., Ren, Y., Chen, F., & Zhao, Z. (2022). Diffsinger: Singing voice synthesis via shallow diffusion mechanism. In Proc. AAAI, 36, 11020\u201311028.","journal-title":"In Proc. AAAI"},{"key":"2634_CR26","unstructured":"Liu, J., Xiong, W., Jones, I., Nie, Y., Gupta, A., & O\u011fuz, B. (2023). Clip-layout: Style-consistent indoor scene synthesis with semantic furniture embedding. arXiv preprint arXiv:2303.03565."},{"key":"2634_CR27","doi-asserted-by":"crossref","unstructured":"Luo, A., Zhang, Z., Wu, J., & Tenenbaum, J. B. (2020). End-to-end optimization of scene layout. In Proc. CVPR, 3753\u20133762.","DOI":"10.1109\/CVPR42600.2020.00381"},{"key":"2634_CR28","first-page":"109202","volume":"37","author":"L Maillard","year":"2024","unstructured":"Maillard, L., Sereyjol-Garros, N., Durand, T., & Ovsjanikov, M. (2024). Debara: Denoising-based 3d room arrangement generation. Advances in Neural Information Processing Systems, 37, 109202\u2013109232.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2634_CR29","doi-asserted-by":"crossref","unstructured":"Maninis, K.-K., Popov, S., Nie\u00dfner, M., & Ferrari, V. (2023). Cad-estate: Large-scale cad model annotation in rgb videos. arXiv preprint arXiv:2306.09011.","DOI":"10.1109\/ICCV51070.2023.01847"},{"key":"2634_CR30","unstructured":"Nichol, A. Q., Dhariwal, P., Ramesh, A., Shyam, P., Mishkin, P., Mcgrew, B., Sutskever, I., & Chen, M. (2022). Glide: Towards photorealistic image generation and editing with text-guided diffusion models. In Proc. ICML, 16784\u201316804."},{"key":"2634_CR31","doi-asserted-by":"crossref","unstructured":"Nie, Y., Dai, A., Han, X., & Nie\u00dfner, M. (2023). Learning 3d scene priors with 2d supervision. In Proc. CVPR, 792\u2013802.","DOI":"10.1109\/CVPR52729.2023.00083"},{"key":"2634_CR32","first-page":"12013","volume":"34","author":"D Paschalidou","year":"2021","unstructured":"Paschalidou, D., Kar, A., Shugrina, M., Kreis, K., Geiger, A., & Fidler, S. (2021). Atiss: Autoregressive transformers for indoor scene synthesis. Advances in Neural Information Processing Systems, 34, 12013\u201312026.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2634_CR33","doi-asserted-by":"crossref","unstructured":"Purkait, P., Zach, C., & Reid, I. (2020). Sg-vae: Scene grammar variational autoencoder to generate new indoor scenes. In Proc. ECCV, 155\u2013171.","DOI":"10.1007\/978-3-030-58586-0_10"},{"key":"2634_CR34","doi-asserted-by":"crossref","unstructured":"Qi, S., Zhu, Y., Huang, S., Jiang, C., & Zhu, S.-C. (2018). Human-centric indoor scene synthesis using stochastic grammar. In Proc. CVPR, 5899\u20135908.","DOI":"10.1109\/CVPR.2018.00618"},{"key":"2634_CR35","unstructured":"Qiu, Z., Yang, Q., Wang, J., Wang, X., Xu, C., Fu, D., Yao, K., Han, J., Ding, E., & Wang, J. (2023). Learning structure-guided diffusion model for 2d human pose estimation. arXiv preprint arXiv:2306.17074."},{"key":"2634_CR36","unstructured":"Radford, A., Kim, J. W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., et al. (2021). Learning transferable visual models from natural language supervision. In Proc. ICML, 8748\u20138763."},{"key":"2634_CR37","doi-asserted-by":"crossref","unstructured":"Ritchie, D., Wang, K., & Lin, Y.-a. (2019). Fast and flexible indoor scene synthesis via deep convolutional generative models. In Proc. CVPR, 6182\u20136190.","DOI":"10.1109\/CVPR.2019.00634"},{"key":"2634_CR38","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., & Ommer, B. (2022). High-resolution image synthesis with latent diffusion models. In Proc. CVPR, 10684\u201310695.","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"2634_CR39","first-page":"11895","volume":"32","author":"Y Song","year":"2019","unstructured":"Song, Y., & Ermon, S. (2019). Generative modeling by estimating gradients of the data distribution. In Proc. NeurIPS, 32, 11895\u201311907.","journal-title":"In Proc. NeurIPS"},{"key":"2634_CR40","first-page":"12438","volume":"33","author":"Y Song","year":"2020","unstructured":"Song, Y., & Ermon, S. (2020). Improved techniques for training score-based generative models. In Proc. NeurIPS, 33, 12438\u201312448.","journal-title":"In Proc. NeurIPS"},{"key":"2634_CR41","first-page":"251","volume":"34","author":"A Szot","year":"2021","unstructured":"Szot, A., Clegg, A., Undersander, E., Wijmans, E., Zhao, Y., Turner, J., Maestre, N., Mukadam, M., Chaplot, D. S., Maksymets, O., et al. (2021). Habitat 2.0: Training home assistants to rearrange their habitat. Advances in Neural Information Processing Systems, 34, 251\u2013266.","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"2","key":"2634_CR42","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/1944846.1944851","volume":"30","author":"JO Talton","year":"2011","unstructured":"Talton, J. O., Lou, Y., Lesser, S., Duke, J., M\u011bch, R., & Koltun, V. (2011). Metropolis procedural modeling. ACM Transactions on Graphics, 30(2), 1\u201314.","journal-title":"ACM Transactions on Graphics"},{"key":"2634_CR43","doi-asserted-by":"crossref","unstructured":"Tang, J., Nie, Y., Markhasin, L., Dai, A., Thies, J., & Nie\u00dfner, M. (2024). Diffuscene: Denoising diffusion models for generative indoor scene synthesis. In Proc. CVPR, 20507\u201320518.","DOI":"10.1109\/CVPR52733.2024.01938"},{"key":"2634_CR44","unstructured":"Tutenel, T., Bidarra, R., Smelik, R. M., & Kraker, K. J. d. (2009). Rule-based layout solving and its application to procedural interior generation. In Proc. 3D Advanced Media In Gaming And Simulation, 15\u201324."},{"issue":"4","key":"2634_CR45","first-page":"1","volume":"38","author":"K Wang","year":"2019","unstructured":"Wang, K., Lin, Y.-A., Weissmann, B., Savva, M., Chang, A. X., & Ritchie, D. (2019). Planit: Planning and instantiating indoor scenes with relation graph and spatial prior networks. ACM Transactions on Graphics, 38(4), 1\u201315.","journal-title":"ACM Transactions on Graphics"},{"issue":"4","key":"2634_CR46","first-page":"1","volume":"37","author":"K Wang","year":"2018","unstructured":"Wang, K., Savva, M., Chang, A. X., & Ritchie, D. (2018). Deep convolutional priors for indoor scene synthesis. ACM Transactions on Graphics, 37(4), 1\u201314.","journal-title":"ACM Transactions on Graphics"},{"key":"2634_CR47","doi-asserted-by":"crossref","unstructured":"Wang, P., Wang, Y., Li, S., Zhang, Z., Lei, Z., & Zhang, L. (2024a). Open vocabulary 3d scene understanding via geometry guided self-distillation. In Proc. ECCV, 442\u2013460. Springer.","DOI":"10.1007\/978-3-031-72633-0_25"},{"key":"2634_CR48","doi-asserted-by":"crossref","unstructured":"Wang, X., Yeshwanth, C., & Nie\u00dfner, M. (2021). Sceneformer: Indoor scene generation with transformers. In Proc. 3DV, 106\u2013115.","DOI":"10.1109\/3DV53792.2021.00021"},{"key":"2634_CR49","doi-asserted-by":"crossref","unstructured":"Wang, Y., Liang, J., & Zhang, Z. (2024b). A curriculum-style self-training approach for source-free semantic segmentation. IEEE Transactions on Pattern Analysis and Machine Intelligence.","DOI":"10.1109\/TPAMI.2024.3432168"},{"key":"2634_CR50","doi-asserted-by":"crossref","unstructured":"Wei, Q. A., Ding, S., Park, J. J., Sajnani, R., Poulenard, A., Sridhar, S., & Guibas, L. (2023). Lego-net: Learning regular rearrangements of objects in rooms. In Proc. CVPR, 19037\u201319047.","DOI":"10.1109\/CVPR52729.2023.01825"},{"key":"2634_CR51","doi-asserted-by":"crossref","unstructured":"Xiao, J., Owens, A., & Torralba, A. (2013). Sun3d: A database of big spaces reconstructed using sfm and object labels. In Proc. ICCV, 1625\u20131632.","DOI":"10.1109\/ICCV.2013.458"},{"key":"2634_CR52","doi-asserted-by":"crossref","unstructured":"Yang, H., Zhang, Z., Yan, S., Huang, H., Ma, C., Zheng, Y., Bajaj, C., & Huang, Q. (2021a). Scene synthesis via uncertainty-driven attribute synchronization. In Proc. ICCV, 5630\u20135640.","DOI":"10.1109\/ICCV48922.2021.00558"},{"key":"2634_CR53","doi-asserted-by":"crossref","unstructured":"Yang, M.-J., Guo, Y.-X., Zhou, B., & Tong, X. (2021b). Indoor scene generation from a collection of semantic-segmented depth images. In Proc. ICCV, 15203\u201315212.","DOI":"10.1109\/ICCV48922.2021.01492"},{"key":"2634_CR54","doi-asserted-by":"crossref","unstructured":"Yang, Z., Lu, K., Zhang, C., Qi, J., Jiang, H., Ma, R., Yin, S., Xu, Y., Xing, M., Xiao, Z., et al. (2025). Mmgdreamer: Mixed-modality graph for geometry-controllable 3d indoor scene generation. arXiv preprint arXiv:2502.05874.","DOI":"10.1609\/aaai.v39i9.33017"},{"key":"2634_CR55","doi-asserted-by":"crossref","unstructured":"Yeshwanth, C., Liu, Y.-C., Nie\u00dfner, M., & Dai, A. (2023). Scannet++: A high-fidelity dataset of 3d indoor scenes. In Proc. ICCV, 12\u201322.","DOI":"10.1109\/ICCV51070.2023.00008"},{"issue":"2","key":"2634_CR56","doi-asserted-by":"publisher","first-page":"1138","DOI":"10.1109\/TVCG.2015.2417575","volume":"22","author":"L-F Yu","year":"2015","unstructured":"Yu, L.-F., Yeung, S.-K., & Terzopoulos, D. (2015). The clutterpalette: An interactive tool for detailing indoor scenes. IEEE Transactions on Visualization and Computer Graphics, 22(2), 1138\u20131148.","journal-title":"IEEE Transactions on Visualization and Computer Graphics"},{"key":"2634_CR57","doi-asserted-by":"crossref","unstructured":"Zhai, G., \u00d6rnek, E. P., Chen, D. Z., Liao, R., Di, Y., Navab, N., Tombari, F., & Busam, B. (2024). Echoscene: Indoor scene generation via information echo over scene graph diffusion. In Proc. ECCV, 167\u2013184. Springer.","DOI":"10.1007\/978-3-031-72664-4_10"},{"key":"2634_CR58","first-page":"30026","volume":"36","author":"G Zhai","year":"2023","unstructured":"Zhai, G., \u00d6rnek, E. P., Wu, S.-C., Di, Y., Tombari, F., Navab, N., & Busam, B. (2023). Commonscenes: Generating commonsense 3d indoor scenes with scene graph diffusion. In Proc. NeurIPS, 36, 30026\u201330038.","journal-title":"In Proc. NeurIPS"},{"key":"2634_CR59","doi-asserted-by":"crossref","unstructured":"Zhang, G., Wang, Y., Luo, C., Xu, S., Ming, Y., Peng, J., & Zhang, M. (2024a). Visual harmony: Llm\u2019s power in crafting coherent indoor scenes from images. In Proc. PRCV, 3\u201317. Springer.","DOI":"10.1007\/978-981-97-8508-7_1"},{"key":"2634_CR60","unstructured":"Zhang, S., Zhou, M., Wang, Y., Luo, C., Wang, R., Li, Y., Zhang, Z., & Peng, J. (2024b). Cityx: Controllable procedural content generation for unbounded 3d cities. arXiv preprint arXiv:2407.17572."},{"key":"2634_CR61","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Chang, Q., Wang, Y., Chen, G., Zhang, Z., & Peng, J. (2025). Emodiffusion: Enhancing emotional 3d facial animation with latent diffusion models. arXiv preprint arXiv:2503.11028.","DOI":"10.2139\/ssrn.5174454"},{"key":"2634_CR62","unstructured":"Zhang, Y., Yang, H., Luo, C., Peng, J., Wang, Y., & Zhang, Z. (2024c). Ood-hoi: Text-driven 3d whole-body human-object interactions generation beyond training domains. arXiv preprint arXiv:2411.18660."},{"issue":"2","key":"2634_CR63","first-page":"1","volume":"39","author":"Z Zhang","year":"2020","unstructured":"Zhang, Z., Yang, Z., Ma, C., Luo, L., Huth, A., Vouga, E., & Huang, Q. (2020). Deep generative modeling for scene synthesis via hybrid representations. ACM Transactions on Graphics, 39(2), 1\u201321.","journal-title":"ACM Transactions on Graphics"},{"key":"2634_CR64","doi-asserted-by":"crossref","unstructured":"Zheng, J., Zhang, J., Li, J., Tang, R., Gao, S., & Zhou, Z. (2020). Structured3d: A large photo-realistic dataset for structured 3d modeling. In Proc. ECCV, 519\u2013535.","DOI":"10.1007\/978-3-030-58545-7_30"},{"key":"2634_CR65","doi-asserted-by":"publisher","first-page":"7641","DOI":"10.1609\/aaai.v38i7.28597","volume":"38","author":"G Zhou","year":"2024","unstructured":"Zhou, G., Hong, Y., & Wu, Q. (2024). Navgpt: Explicit reasoning in vision-and-language navigation with large language models. In Proc. AAAI, 38, 7641\u20137649.","journal-title":"In Proc. AAAI"},{"key":"2634_CR66","doi-asserted-by":"publisher","first-page":"10806","DOI":"10.1609\/aaai.v39i10.33174","volume":"39","author":"M Zhou","year":"2025","unstructured":"Zhou, M., Wang, Y., Hou, J., Zhang, S., Li, Y., Luo, C., Peng, J., & Zhang, Z. (2025). Scenex: Procedural controllable large-scale scene generation. In Proc. AAAI, 39, 10806\u201310814.","journal-title":"In Proc. AAAI"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02634-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-025-02634-w","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02634-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,27]],"date-time":"2026-03-27T08:39:25Z","timestamp":1774600765000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-025-02634-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2,21]]},"references-count":66,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2026,3]]}},"alternative-id":["2634"],"URL":"https:\/\/doi.org\/10.1007\/s11263-025-02634-w","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,2,21]]},"assertion":[{"value":"12 September 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 October 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 February 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"The proposed dataset\n                      FurniScene\n                      will be made publicly available if the paper is accepted, which aims to make some contributions to the community of 3D indoor scene generation and embodied intelligence.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declaration"}}],"article-number":"125"}}