{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T11:29:34Z","timestamp":1764588574893,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":48,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681699","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:33Z","timestamp":1729925973000},"page":"10182-10190","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Context-Aware Indoor Point Cloud Object Generation through User Instructions"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-8094-180X","authenticated-orcid":false,"given":"Luo","family":"Yiyang","sequence":"first","affiliation":[{"name":"Nanyang Technological University, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-5376-7881","authenticated-orcid":false,"given":"Ke","family":"Lin","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-6321-0475","authenticated-orcid":false,"given":"Chao","family":"Gu","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_25"},{"key":"e_1_3_2_1_2_1","volume-title":"Int. Conf. Mach. Learn. (10--15","volume":"80","author":"Achlioptas P.","year":"2018","unstructured":"Achlioptas, P., Diamanti, O., Mitliagkas, I., and Guibas, L. Learning representations and generative models for 3D point clouds. In Int. Conf. Mach. Learn. (10--15 Jul 2018), J. Dy and A. Krause, Eds., vol. 80 of Proceedings of Machine Learning Research, pp. 40--49."},{"key":"e_1_3_2_1_3_1","first-page":"25102","article-title":"Gaudi: A neural architect for immersive 3d scene generation","volume":"35","author":"Bautista M. A.","year":"2022","unstructured":"Bautista, M. A., Guo, P., Abnar, S., Talbott,W., Toshev, A., Chen, Z., Dinh, L., Zhai, S., Goh, H., Ulbricht, D., et al. Gaudi: A neural architect for immersive 3d scene generation. NeurIPS 35 (2022), 25102--25116.","journal-title":"NeurIPS"},{"key":"e_1_3_2_1_4_1","first-page":"1877","volume-title":"NeurIPS","volume":"33","author":"Brown T.","year":"2020","unstructured":"Brown, T., Mann, B., Ryder, N., Subbiah, M., Kaplan, J. D., Dhariwal, P., Neelakantan, A., Shyam, P., Sastry, G., Askell, A., et al. Language models are few-shot learners. In NeurIPS (2020), vol. 33, pp. 1877--1901."},{"key":"e_1_3_2_1_5_1","volume-title":"Shapenet: An information-rich 3d model repository. arXiv preprint arXiv:1512.03012","author":"Chang A. X.","year":"2015","unstructured":"Chang, A. X., Funkhouser, T., Guibas, L., Hanrahan, P., Huang, Q., Li, Z., Savarese, S., Savva, M., Song, S., Su, H., et al. Shapenet: An information-rich 3d model repository. arXiv preprint arXiv:1512.03012 (2015)."},{"key":"e_1_3_2_1_6_1","volume-title":"Scanrefer: 3d object localization in rgb-d scans using natural language. ECCV","author":"Chen D. Z.","year":"2020","unstructured":"Chen, D. Z., Chang, A. X., and Niessner, M. Scanrefer: 3d object localization in rgb-d scans using natural language. ECCV (2020)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-20893-6_7"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01070"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00321"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.261"},{"key":"e_1_3_2_1_11_1","volume-title":"Quickfps: Architecture and algorithm co-design for farthest point sampling in large-scale point clouds","author":"Han M.","year":"2023","unstructured":"Han, M.,Wang, L., Xiao, L., Zhang, H., Zhang, C., Xu, X., and Zhu, J. Quickfps: Architecture and algorithm co-design for farthest point sampling in large-scale point clouds. IEEE Trans. Computer-Aided Des. Integr. Circuits and Syst. (2023)."},{"key":"e_1_3_2_1_12_1","first-page":"6840","article-title":"Denoising diffusion probabilistic models","volume":"33","author":"Ho J.","year":"2020","unstructured":"Ho, J., Jain, A., and Abbeel, P. Denoising diffusion probabilistic models. NeurIPS 33 (2020), 6840--6851.","journal-title":"NeurIPS"},{"key":"e_1_3_2_1_13_1","volume-title":"Classifier-free diffusion guidance. arXiv preprint arXiv:2207.12598","author":"Ho J.","year":"2022","unstructured":"Ho, J., and Salimans, T. Classifier-free diffusion guidance. arXiv preprint arXiv:2207.12598 (2022)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00727"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01508"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3611902"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58555-6_24"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19833-5_31"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01041"},{"key":"e_1_3_2_1_20_1","first-page":"4171","volume-title":"NAACL","author":"Kenton J. D. M.-W. C.","year":"2019","unstructured":"Kenton, J. D. M.-W. C., and Toutanova, L. K. Bert: Pre-training of deep bidirectional transformers for language understanding. In NAACL (2019), pp. 4171--4186."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00059"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCE.2022.3141093"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01737"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00286"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01242"},{"key":"e_1_3_2_1_26_1","first-page":"1","article-title":"Nerf: Representing scenes as neural radiance fields for view synthesis","volume":"65","author":"Mildenhall B.","year":"2021","unstructured":"Mildenhall, B., Srinivasan, P. P., Tancik, M., Barron, J. T., Ramamoorthi, R., and Ng, R. Nerf: Representing scenes as neural radiance fields for view synthesis. Commun. ACM 65, 1 (2021), 99--106.","journal-title":"Commun. ACM"},{"key":"e_1_3_2_1_27_1","volume-title":"Point-e: A system for generating 3d point clouds from complex prompts. arXiv preprint arXiv:2212.08751","author":"Nichol A.","year":"2022","unstructured":"Nichol, A., Jun, H., Dhariwal, P., Mishkin, P., and Chen, M. Point-e: A system for generating 3d point clouds from complex prompts. arXiv preprint arXiv:2212.08751 (2022)."},{"key":"e_1_3_2_1_28_1","volume-title":"Gpt-4 technical report. arXiv preprint arXiv:2303.08774","author":"Open AI.","year":"2023","unstructured":"OpenAI. Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_29_1","first-page":"12013","article-title":"Atiss: Autoregressive transformers for indoor scene synthesis","volume":"34","author":"Paschalidou D.","year":"2021","unstructured":"Paschalidou, D., Kar, A., Shugrina, M., Kreis, K., Geiger, A., and Fidler, S. Atiss: Autoregressive transformers for indoor scene synthesis. NeurIPS 34 (2021), 12013--12026.","journal-title":"NeurIPS"},{"key":"e_1_3_2_1_30_1","first-page":"30","article-title":"Pointnet: Deep hierarchical feature learning on point sets in a metric space","author":"Qi C. R.","year":"2017","unstructured":"Qi, C. R., Yi, L., Su, H., and Guibas, L. J. Pointnet: Deep hierarchical feature learning on point sets in a metric space. NeurIPS 30 (2017).","journal-title":"NeurIPS"},{"key":"e_1_3_2_1_31_1","volume-title":"NeurIPS","author":"Qian G.","year":"2022","unstructured":"Qian, G., Li, Y., Peng, H., Mai, J., Hammoud, H., Elhoseiny, M., and Ghanem, B. Pointnext: Revisiting pointnet with improved training and scaling strategies. In NeurIPS (2022)."},{"key":"e_1_3_2_1_32_1","volume-title":"Int. Conf. Mach. Learn. (18--24","volume":"139","author":"Radford A.","year":"2021","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., Krueger, G., and Sutskever, I. Learning transferable visual models from natural language supervision. In Int. Conf. Mach. Learn. (18--24 Jul 2021), M. Meila and T. Zhang, Eds., vol. 139 of Proceedings of Machine Learning Research, pp. 8748--8763."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00355"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA46639.2022.9811816"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00634"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1023\/A:1026543900054"},{"key":"e_1_3_2_1_37_1","volume-title":"Int. Conf. Mach. Learn.","author":"Sohl-Dickstein J.","year":"2015","unstructured":"Sohl-Dickstein, J., Weiss, E., Maheswaranathan, N., and Ganguli, S. Deep unsupervised learning using nonequilibrium thermodynamics. In Int. Conf. Mach. Learn. (Lille, France, 07--09 Jul 2015), F. Bach and D. Blei, Eds., vol. 37 of Proceedings of Machine Learning Research, pp. 2256--2265."},{"key":"e_1_3_2_1_38_1","volume-title":"Roomdreamer: Text-driven 3d indoor scene synthesis with coherent geometry and texture. arXiv preprint arXiv:2305.11337","author":"Song L.","year":"2023","unstructured":"Song, L., Cao, L., Xu, H., Kang, K., Tang, F., Yuan, J., and Zhao, Y. Roomdreamer: Text-driven 3d indoor scene synthesis with coherent geometry and texture. arXiv preprint arXiv:2305.11337 (2023)."},{"key":"e_1_3_2_1_39_1","volume-title":"I. Attention is all you need. In NeurIPS","author":"Vaswani A.","year":"2017","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A. N., Kaiser, L. u., and Polosukhin, I. Attention is all you need. In NeurIPS (2017), I. Guyon, U. V. Luxburg, S. Bengio, H. Wallach, R. Fergus, S. Vishwanathan, and R. Garnett, Eds., vol. 30."},{"key":"e_1_3_2_1_40_1","first-page":"106","volume-title":"Sceneformer: Indoor scene generation with transformers. In 3DV","author":"Wang X.","year":"2021","unstructured":"Wang, X., Yeshwanth, C., and Niessner, M. Sceneformer: Indoor scene generation with transformers. In 3DV (2021), IEEE, pp. 106--115."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02003"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612262"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612226"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00464"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00181"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01397"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00577"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00748"}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Melbourne VIC Australia","acronym":"MM '24"},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681699","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681699","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:50Z","timestamp":1750295870000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681699"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":48,"alternative-id":["10.1145\/3664647.3681699","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681699","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}