{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T08:57:51Z","timestamp":1785488271683,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":20,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,12,17]],"date-time":"2025-12-17T00:00:00Z","timestamp":1765929600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,17]]},"DOI":"10.1145\/3774521.3774535","type":"proceedings-article","created":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T07:34:24Z","timestamp":1785483264000},"page":"1-9","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Screen2UI: Pixel-Only On-Device UI Hierarchy Generation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-5091-0897","authenticated-orcid":false,"given":"Anup","family":"Kushwaha","sequence":"first","affiliation":[{"name":"Samsung R&amp;D Institute Bangalore, Bengaluru, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-7787-9186","authenticated-orcid":false,"given":"Banseedhar","family":"Gondaliya","sequence":"additional","affiliation":[{"name":"Samsung R&amp;D Institute Bangalore, Bengaluru, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-9348-1685","authenticated-orcid":false,"given":"Chandramouli","family":"Sanchi","sequence":"additional","affiliation":[{"name":"Samsung R&amp;D Institute Bangalore, Bengaluru, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-0339-6386","authenticated-orcid":false,"given":"Surya","family":"Kumar","sequence":"additional","affiliation":[{"name":"Samsung R&amp;D Institute Bangalore, Bengaluru, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-3306-5125","authenticated-orcid":false,"given":"Apil","family":"Thapa","sequence":"additional","affiliation":[{"name":"Samsung R&amp;D Institute Bangalore, Bengaluru, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-2016-7553","authenticated-orcid":false,"given":"Edula Sri Ranga Srihit","family":"Reddy","sequence":"additional","affiliation":[{"name":"Samsung R&amp;D Institute Bangalore, Bengaluru, India"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,31]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445762"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3502073"},{"key":"e_1_3_3_1_4_2","unstructured":"Seyed\u00a0Shayan Daneshvar and Shaowei Wang. 2024. GUI Element Detection Using SOTA YOLO Deep Learning Models. arxiv:https:\/\/arXiv.org\/abs\/2408.03507\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2408.03507"},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"publisher","DOI":"10.1145\/3126594.3126651"},{"key":"e_1_3_3_1_6_2","unstructured":"Longxi Gao Li Zhang Shihe Wang Shangguang Wang Yuanchun Li and Mengwei Xu. 2024. Mobileviews: A large-scale mobile gui dataset. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.14337 (2024)."},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","DOI":"10.1145\/3708359.3712129"},{"key":"e_1_3_3_1_9_2","volume-title":"Ultralytics YOLO11","author":"Jocher Glenn","year":"2024","unstructured":"Glenn Jocher and Jing Qiu. 2024. Ultralytics YOLO11. https:\/\/github.com\/ultralytics\/ultralytics"},{"key":"e_1_3_3_1_10_2","unstructured":"Gang Li and Yang Li. 2022. Spotlight: Mobile ui understanding using vision-language models with a focus. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2209.14927 (2022)."},{"key":"e_1_3_3_1_11_2","series-title":"(NIPS \u201920)","volume-title":"Proceedings of the 34th International Conference on Neural Information Processing Systems","author":"Li Xiang","year":"2020","unstructured":"Xiang Li, Wenhai Wang, Lijun Wu, Shuo Chen, Xiaolin Hu, Jun Li, Jinhui Tang, and Jian Yang. 2020. Generalized focal loss: learning qualified and distributed bounding boxes for dense object detection. In Proceedings of the 34th International Conference on Neural Information Processing Systems (Vancouver, BC, Canada) (NIPS \u201920). Curran Associates Inc., Red Hook, NY, USA, Article 1763, 11\u00a0pages."},{"key":"e_1_3_3_1_12_2","unstructured":"Yang Li Gang Li Xin Zhou Mostafa Dehghani and Alexey Gritsenko. 2021. Vut: Versatile ui transformer for multi-modal multi-task user interface modeling. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2112.05692 (2021)."},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"e_1_3_3_1_14_2","first-page":"466","volume-title":"European Conference on Computer Vision","author":"Peng Yi-Hao","year":"2024","unstructured":"Yi-Hao Peng, Faria Huq, Yue Jiang, Jason Wu, Xin\u00a0Yue Li, Jeffrey\u00a0P Bigham, and Amy Pavel. 2024. Dreamstruct: Understanding slides and user interfaces via synthetic data generation. In European Conference on Computer Vision. Springer, 466\u2013485."},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00474"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3517497"},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300305"},{"key":"e_1_3_3_1_18_2","first-page":"10096","volume-title":"International conference on machine learning","author":"Tan Mingxing","year":"2021","unstructured":"Mingxing Tan and Quoc Le. 2021. Efficientnetv2: Smaller models and faster training. In International conference on machine learning. PMLR, 10096\u201310106."},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"publisher","DOI":"10.1145\/3472749.3474763"},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445186"},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6999"}],"event":{"name":"ICVGIP 2025: Indian Conference on Computer Vision, Graphics, and Image Processing","location":"Mandi Himachal Pradesh India","acronym":"ICVGIP 2025"},"container-title":["Proceedings of the Sixteen Indian Conference on Computer Vision, Graphics and Image Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774521.3774535","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T08:02:31Z","timestamp":1785484951000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774521.3774535"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,17]]},"references-count":20,"alternative-id":["10.1145\/3774521.3774535","10.1145\/3774521"],"URL":"https:\/\/doi.org\/10.1145\/3774521.3774535","relation":{},"subject":[],"published":{"date-parts":[[2025,12,17]]},"assertion":[{"value":"2026-07-31","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}