{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T19:06:17Z","timestamp":1784228777047,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":47,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,19]]},"DOI":"10.1145\/3799902.3811124","type":"proceedings-article","created":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T16:15:27Z","timestamp":1784218527000},"page":"1-11","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Raster2Seq: Polygon Sequence Generation for Floorplan Reconstruction"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6834-4139","authenticated-orcid":false,"given":"Hao","family":"Phung","sequence":"first","affiliation":[{"name":"Computer Science, Cornell University, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3476-0940","authenticated-orcid":false,"given":"Hadar","family":"Averbuch-Elor","sequence":"additional","affiliation":[{"name":"Computer Science, Cornell University, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_3_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00096"},{"key":"e_1_3_3_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2011.177"},{"key":"e_1_3_3_2_4_1","first-page":"247","volume-title":"European Conference on Computer Vision","author":"Avetisyan Armen","year":"2024","unstructured":"Armen Avetisyan, Christopher Xie, Henry Howard-Jenkins, Tsun-Yi Yang, Samir Aroudj, Suvam Patra, Fuyang Zhang, Duncan Frost, Luke Holland, Campbell Orme, et\u00a0al. 2024. Scenescript: Reconstructing scenes with an autoregressive structured language model. In European Conference on Computer Vision. Springer, 247\u2013263."},{"key":"e_1_3_3_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.546"},{"key":"e_1_3_3_2_6_1","doi-asserted-by":"crossref","unstructured":"Jiacheng Chen Ruizhi Deng and Yasutaka Furukawa. 2023. Polydiffuse: Polygonal shape reconstruction via guided set diffusion models. Advances in Neural Information Processing Systems 36 (2023) 1863\u20131888.","DOI":"10.52202\/075280-0090"},{"key":"e_1_3_3_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00275"},{"key":"e_1_3_3_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00384"},{"key":"e_1_3_3_2_9_1","unstructured":"Ting Chen Saurabh Saxena Lala Li David\u00a0J Fleet and Geoffrey Hinton. 2021. Pix2seq: A language modeling framework for object detection. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2109.10852 (2021)."},{"key":"e_1_3_3_2_10_1","doi-asserted-by":"crossref","unstructured":"Ting Chen Saurabh Saxena Lala Li Tsung-Yi Lin David\u00a0J Fleet and Geoffrey\u00a0E Hinton. 2022b. A unified sequence interface for vision tasks. Advances in Neural Information Processing Systems 35 (2022) 31333\u201331346.","DOI":"10.52202\/068431-2272"},{"key":"e_1_3_3_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01059"},{"key":"e_1_3_3_2_12_1","doi-asserted-by":"crossref","unstructured":"Lluis-Pere De\u00a0Las\u00a0Heras Sheraz Ahmed Marcus Liwicki Ernest Valveny and Gemma S\u00e1nchez. 2014. Statistical segmentation and structural recognition for floor plan interpretation: Notation invariant structural element recognition. International Journal on Document Analysis and Recognition (IJDAR) 17 3 (2014) 221\u2013237.","DOI":"10.1007\/s10032-013-0215-2"},{"key":"e_1_3_3_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV61041.2025.00152"},{"key":"e_1_3_3_2_14_1","doi-asserted-by":"publisher","DOI":"10.1111\/cgf.15007"},{"key":"e_1_3_3_2_15_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-20205-7_3"},{"key":"e_1_3_3_2_16_1","doi-asserted-by":"crossref","unstructured":"Jun Li Kai Xu Siddhartha Chaudhuri Ersin Yumer Hao Zhang and Leonidas Guibas. 2017. Grass: Generative recursive autoencoders for shape structures. ACM Transactions on Graphics (TOG) 36 4 (2017) 1\u201314.","DOI":"10.1145\/3072959.3073637"},{"key":"e_1_3_3_2_17_1","doi-asserted-by":"crossref","unstructured":"Manyi Li Akshay\u00a0Gadi Patil Kai Xu Siddhartha Chaudhuri Owais Khan Ariel Shamir Changhe Tu Baoquan Chen Daniel Cohen-Or and Hao Zhang. 2019. Grains: Generative recursive autoencoders for indoor scenes. ACM Transactions on Graphics (TOG) 38 2 (2019) 1\u201316.","DOI":"10.1145\/3303766"},{"key":"e_1_3_3_2_18_1","doi-asserted-by":"crossref","unstructured":"Tianhong Li Yonglong Tian He Li Mingyang Deng and Kaiming He. 2024. Autoregressive image generation without vector quantization. Advances in Neural Information Processing Systems 37 (2024) 56424\u201356445.","DOI":"10.52202\/079017-1797"},{"key":"e_1_3_3_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.324"},{"key":"e_1_3_3_2_20_1","first-page":"3413","volume-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","author":"Liu Chenxi","year":"2015","unstructured":"Chenxi Liu, Alexander\u00a0G Schwing, Kaustav Kundu, Raquel Urtasun, and Sanja Fidler. 2015. Rent3d: Floor-plan priors for monocular layout estimation. In Proceedings of the IEEE conference on computer vision and pattern recognition. 3413\u20133421."},{"key":"e_1_3_3_2_21_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01231-1_13"},{"key":"e_1_3_3_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.241"},{"key":"e_1_3_3_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01789"},{"key":"e_1_3_3_2_24_1","volume-title":"European Conference on Computer Vision","author":"Liu Yuzhou","year":"2024","unstructured":"Yuzhou Liu, Lingjie Zhu, Xiaodong Ma, Hanqiao Ye, Xiang Gao, Xianwei Zheng, and Shuhan Shen. 2024. PolyRoom: Room-aware Transformer for Floorplan Reconstruction. In European Conference on Computer Vision."},{"key":"e_1_3_3_2_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/1815330.1815352"},{"key":"e_1_3_3_2_26_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10578-9_1"},{"key":"e_1_3_3_2_27_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58523-5_30"},{"key":"e_1_3_3_2_28_1","unstructured":"Hieu\u00a0T Nguyen Yiwen Chen Vikram Voleti Varun Jampani and Huaizu Jiang. 2024. HouseCrafter: Lifting Floorplans to 3D Scenes with 2D Diffusion Model. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.20077 (2024)."},{"key":"e_1_3_3_2_29_1","unstructured":"Despoina Paschalidou Amlan Kar Maria Shugrina Karsten Kreis Andreas Geiger and Sanja Fidler. 2021. Atiss: Autoregressive transformers for indoor scene synthesis. Advances in Neural Information Processing Systems 34 (2021) 12013\u201312026."},{"key":"e_1_3_3_2_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW50498.2020.00280"},{"key":"e_1_3_3_2_31_1","first-page":"8821","volume-title":"International conference on machine learning","author":"Ramesh Aditya","year":"2021","unstructured":"Aditya Ramesh, Mikhail Pavlov, Gabriel Goh, Scott Gray, Chelsea Voss, Alec Radford, Mark Chen, and Ilya Sutskever. 2021. Zero-shot text-to-image generation. In International conference on machine learning. Pmlr, 8821\u20138831."},{"key":"e_1_3_3_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00529"},{"key":"e_1_3_3_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00413"},{"key":"e_1_3_3_2_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01573"},{"key":"e_1_3_3_2_35_1","doi-asserted-by":"crossref","unstructured":"Jiahui Sun Wenming Wu Ligang Liu Wenjie Min Gaofeng Zhang and Liping Zheng. 2022. Wallplan: synthesizing floorplans by learning to generate wall graphs. ACM Transactions on Graphics (TOG) 41 4 (2022) 1\u201314.","DOI":"10.1145\/3528223.3530135"},{"key":"e_1_3_3_2_36_1","unstructured":"Ilya Sutskever Oriol Vinyals and Quoc\u00a0V Le. 2014. Sequence to sequence learning with neural networks. Advances in neural information processing systems 27 (2014)."},{"key":"e_1_3_3_2_37_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan\u00a0N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_3_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"e_1_3_3_2_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.309"},{"key":"e_1_3_3_2_40_1","volume-title":"ECCV","author":"Xu Honghao","year":"2024","unstructured":"Honghao Xu, Juzhan Xu, Zeyu Huang, Pengfei Xu, Hui Huang, and Ruizhen Hu. 2024. FRI-Net: Floorplan Reconstruction via Room-wise Implicit Representation. In ECCV."},{"key":"e_1_3_3_2_41_1","first-page":"2048","volume-title":"International conference on machine learning","author":"Xu Kelvin","year":"2015","unstructured":"Kelvin Xu, Jimmy Ba, Ryan Kiros, Kyunghyun Cho, Aaron Courville, Ruslan Salakhudinov, Rich Zemel, and Yoshua Bengio. 2015. Show, attend and tell: Neural image caption generation with visual attention. In International conference on machine learning. PMLR, 2048\u20132057."},{"key":"e_1_3_3_2_42_1","unstructured":"Jiahui Yu Yuanzhong Xu Jing\u00a0Yu Koh Thang Luong Gunjan Baid Zirui Wang Vijay Vasudevan Alexander Ku Yinfei Yang Burcu\u00a0Karagol Ayan Ben Hutchinson Wei Han Zarana Parekh Xin Li Han Zhang Jason Baldridge and Yonghui Wu. 2022. Scaling Autoregressive Models for Content-Rich Text-to-Image Generation. Transactions on Machine Learning Research (2022). https:\/\/openreview.net\/forum?id=AFDcYJKhND Featured Certification."},{"key":"e_1_3_3_2_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00088"},{"key":"e_1_3_3_2_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00919"},{"key":"e_1_3_3_2_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00978"},{"key":"e_1_3_3_2_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3680798"},{"key":"e_1_3_3_2_47_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58545-7_30"},{"key":"e_1_3_3_2_48_1","volume-title":"International Conference on Learning Representations","author":"Zhu Xizhou","year":"2021","unstructured":"Xizhou Zhu, Weijie Su, Lewei Lu, Bin Li, Xiaogang Wang, and Jifeng Dai. 2021. Deformable {DETR}: Deformable Transformers for End-to-End Object Detection. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=gZ9hCDWe6ke"}],"event":{"name":"SIGGRAPH Conference Papers '26: Special Interest Group on Computer Graphics and Interactive Techniques Conference Conference Papers","location":"Los Angeles CA USA","acronym":"SIGGRAPH Conference Papers '26","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Proceedings of the Special Interest Group on Computer Graphics and Interactive Techniques Conference Conference Papers"],"original-title":[],"deposited":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T18:26:15Z","timestamp":1784226375000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3799902.3811124"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":47,"alternative-id":["10.1145\/3799902.3811124","10.1145\/3799902"],"URL":"https:\/\/doi.org\/10.1145\/3799902.3811124","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}