{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T08:01:35Z","timestamp":1776931295711,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":116,"publisher":"ACM","funder":[{"name":"ERC Advanced Grant","award":["101141916"],"award-info":[{"award-number":["101141916"]}]},{"name":"Research Council of Finland (FCAI)","award":["328400"],"award-info":[{"award-number":["328400"]}]},{"name":"Research Council of Finland (FCAI)","award":["345604"],"award-info":[{"award-number":["345604"]}]},{"name":"Research Council of Finland (FCAI)","award":["341763"],"award-info":[{"award-number":["341763"]}]},{"name":"Research Council of Finland (Subjective Functions)","award":["357578"],"award-info":[{"award-number":["357578"]}]},{"name":"European Innovation Council (SYMBIOTIK project)","award":["101071147"],"award-info":[{"award-number":["101071147"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,13]]},"DOI":"10.1145\/3772318.3791178","type":"proceedings-article","created":{"date-parts":[[2026,4,13]],"date-time":"2026-04-13T04:12:36Z","timestamp":1776053556000},"page":"1-23","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["SeekUI: Predicting Visual Search Behavior on Graphical User Interfaces with a Reward-Augmented Vision Language Model"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7088-2331","authenticated-orcid":false,"given":"Zixin","family":"Guo","sequence":"first","affiliation":[{"name":"Aalto University, Espoo, Finland"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0022-6512","authenticated-orcid":false,"given":"Yue","family":"Jiang","sequence":"additional","affiliation":[{"name":"Department of Information and Communications Engineering, Aalto University, Espoo, Finland and Kahlert School of Computing, University of Utah, Salt Lake City, Utah, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5011-1847","authenticated-orcid":false,"given":"Luis A.","family":"Leiva","sequence":"additional","affiliation":[{"name":"University of Luxembourg, Esch-sur-Alzette, Luxembourg"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2498-7837","authenticated-orcid":false,"given":"Antti","family":"Oulasvirta","sequence":"additional","affiliation":[{"name":"Department of Information and Communications Engineering, Aalto University, Helsinki, Finland and ELLIS Institute Finland, Helsinki, Finland"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,4,13]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"crossref","unstructured":"Jean-Baptiste Alayrac Jeff Donahue Pauline Luc Antoine Miech Iain Barr Yana Hasson Karel Lenc Arthur Mensch Katie Millican Malcolm Reynolds Roman Ring Eliza Rutherford Serkan Cabi Tengda Han Zhitao Gong Sina Samangooei Marianne Monteiro Jacob Menick Sebastian Borgeaud Andrew Brock Aida Nematzadeh Sahand Sharifzadeh Mikolaj Binkowski Ricardo Barreira Oriol Vinyals Andrew Zisserman and Karen Simonyan. 2022. Flamingo: a Visual Language Model for Few-Shot Learning. arxiv:https:\/\/arXiv.org\/abs\/2204.14198\u00a0[cs.CV]","DOI":"10.52202\/068431-1723"},{"key":"e_1_3_3_2_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2017.275"},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"crossref","unstructured":"Marc Assens Xavier\u00a0Giro i Nieto Kevin McGuinness and Noel\u00a0E. O\u2019Connor. 2018. PathGAN: Visual Scanpath Prediction with Generative Adversarial Networks. ECCV Workshop on Egocentric Perception Interaction and Computing (EPIC).","DOI":"10.1007\/978-3-030-11021-5_25"},{"key":"e_1_3_3_2_5_2","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Sibo Song Kai Dang Peng Wang Shijie Wang Jun Tang Humen Zhong Yuanzhi Zhu Mingkun Yang Zhaohai Li Jianqiang Wan Pengfei Wang Wei Ding Zheren Fu Yiheng Xu Jiabo Ye Xi Zhang Tianbao Xie Zesen Cheng Hang Zhang Zhibo Yang Haiyang Xu and Junyang Lin. 2025. Qwen2.5-VL Technical Report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2502.13923 (2025)."},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"publisher","DOI":"10.1145\/2556288.2557093"},{"key":"e_1_3_3_2_7_2","first-page":"65","volume-title":"Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization","author":"Banerjee Satanjeev","year":"2005","unstructured":"Satanjeev Banerjee and Alon Lavie. 2005. METEOR: An automatic metric for MT evaluation with improved correlation with human judgments. In Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization. 65\u201372."},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"publisher","DOI":"10.1145\/3313831.3376849"},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445519"},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"crossref","unstructured":"Stephan\u00a0A Brandt and Lawrence\u00a0W Stark. 1997. Spontaneous eye movements during visual imagery reflect the content of the visual scene. Journal of cognitive neuroscience 9 1 (1997) 27\u201338.","DOI":"10.1162\/jocn.1997.9.1.27"},{"key":"e_1_3_3_2_11_2","doi-asserted-by":"publisher","DOI":"10.1145\/2556288.2557064"},{"key":"e_1_3_3_2_12_2","unstructured":"Zoya Bylinskii Tilke Judd Ali Borji Laurent Itti Fr\u00e9do Durand Aude Oliva and Antonio Torralba. 2015. Mit saliency benchmark. (2015)."},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"publisher","DOI":"10.1145\/169059.169369"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01073"},{"key":"e_1_3_3_2_15_2","first-page":"314","volume-title":"European Conference on Computer Vision","author":"Chen Xianyu","year":"2024","unstructured":"Xianyu Chen, Ming Jiang, and Qi Zhao. 2024. Gazexplain: Learning to predict natural language explanations of visual scanpaths. In European Conference on Computer Vision. Springer, 314\u2013333."},{"key":"e_1_3_3_2_16_2","unstructured":"Xi Chen Xiao Wang Soravit Changpinyo AJ Piergiovanni Piotr Padlewski Daniel Salz Sebastian Goodman Adam Grycner Basil Mustafa Lucas Beyer Alexander Kolesnikov Joan Puigcerver Nan Ding Keran Rong Hassan Akbari Gaurav Mishra Linting Xue Ashish Thapliyal James Bradbury Weicheng Kuo Mojtaba Seyedhosseini Chao Jia Burcu\u00a0Karagol Ayan Carlos Riquelme Andreas Steiner Anelia Angelova Xiaohua Zhai Neil Houlsby and Radu Soricut. 2023. PaLI: A Jointly-Scaled Multilingual Language-Image Model. arxiv:https:\/\/arXiv.org\/abs\/2209.06794\u00a0[cs.CV]"},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"publisher","unstructured":"Xin Chen and G.J. Zelinsky. 2006. Real-world visual search is dominated by top-down guidance. Vision Research 46 24 (2006) 4118\u20134133. 10.1016\/j.visres.2006.08.008","DOI":"10.1016\/j.visres.2006.08.008"},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"crossref","unstructured":"Zhenzhong Chen and Wanjie Sun. 2018. Scanpath Prediction for Visual Attention Using IOR-ROI LSTM(IJCAI\u201918). AAAI Press 642\u2013648.","DOI":"10.24963\/ijcai.2018\/89"},{"key":"e_1_3_3_2_19_2","unstructured":"Wei-Lin Chiang Zhuohan Li Zi Lin Ying Sheng Zhanghao Wu Hao Zhang Lianmin Zheng Siyuan Zhuang Yonghao Zhuang Joseph\u00a0E Gonzalez et\u00a0al. 2023. Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality. See https:\/\/vicuna.lmsys.org (accessed 14 April 2023) (2023)."},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"crossref","unstructured":"Filipe Cristino Sebastiaan Math\u00f4t Jan Theeuwes and Iain\u00a0D Gilchrist. 2010. ScanMatch: A novel method for comparing fixation sequences. Behavior research methods 42 3 (2010) 692\u2013700.","DOI":"10.3758\/BRM.42.3.692"},{"key":"e_1_3_3_2_21_2","unstructured":"Wenliang Dai Junnan Li Dongxu Li Anthony Meng\u00a0Huat Tiong Junqi Zhao Weisheng Wang Boyang Li Pascale Fung and Steven Hoi. 2023. InstructBLIP: Towards General-purpose Vision-Language Models with Instruction Tuning. arxiv:https:\/\/arXiv.org\/abs\/2305.06500\u00a0[cs.CV]"},{"key":"e_1_3_3_2_22_2","doi-asserted-by":"crossref","unstructured":"Richard Dewhurst Marcus Nystr\u00f6m Halszka Jarodzka Tom Foulsham Roger Johansson and Kenneth Holmqvist. 2012. It depends on how you look at it: Scanpath comparison in multiple dimensions with MultiMatch a vector-based approach. Behavior research methods 44 (2012) 1079\u20131100.","DOI":"10.3758\/s13428-012-0212-2"},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"publisher","DOI":"10.1145\/3025453.3025834"},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"crossref","unstructured":"Parvin Emami Yue Jiang Zixin Guo and Luis\u00a0A Leiva. 2024. Impact of Design Decisions in Scanpath Modeling. Proceedings of the ACM on Human-Computer Interaction 8 ETRA (2024) 1\u201316.","DOI":"10.1145\/3655602"},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"publisher","DOI":"10.1145\/985692.985780"},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"crossref","unstructured":"Tom Foulsham and Geoffrey Underwood. 2008. What can saliency models predict about eye movements? Spatial and sequential aspects of fixations during encoding and recognition. Journal of vision 8 2 (2008) 6\u20136.","DOI":"10.1167\/8.2.6"},{"key":"e_1_3_3_2_27_2","unstructured":"Aryan Garg Yue Jiang and Antti Oulasvirta. 2025. Controllable gui exploration. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2502.03330 (2025)."},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"crossref","unstructured":"Melvyn\u00a0A Goodale and A\u00a0David Milner. 1992. Separate visual pathways for perception and action. Trends in neurosciences 15 1 (1992) 20\u201325.","DOI":"10.1016\/0166-2236(92)90344-8"},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"publisher","unstructured":"Michael Grahame Jason Laberge and C.T. Scialfa. 2004. Age Differences in Search of Web Pages: The Effects of Link Size Link Number and Clutter. Human Factors 46 3 (2004) 385\u2013398. 10.1518\/hfes.46.3.385.50404PMID 15573540.","DOI":"10.1518\/hfes.46.3.385.50404"},{"key":"e_1_3_3_2_30_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-emnlp.537"},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.aacl-short.5"},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"publisher","DOI":"10.1145\/3539618.3592038"},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"publisher","unstructured":"T. Halverson and A.J. Hornof. 2004. Local Density Guides Visual Search: Sparse Groups are First and Faster. Proceedings of the Human Factors and Ergonomics Society Annual Meeting 48 16 (2004) 1860\u20131864. 10.1177\/154193120404801615","DOI":"10.1177\/154193120404801615"},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"publisher","unstructured":"Tim Halverson and A.J. Hornof. 2011. A Computational Model of \u201cActive Vision\u201d for Visual Search in Human\u2013Computer Interaction. Human\u2013Computer Interaction 26 4 (2011) 285\u2013314. 10.1080\/07370024.2011.625237","DOI":"10.1080\/07370024.2011.625237"},{"key":"e_1_3_3_2_35_2","doi-asserted-by":"publisher","DOI":"10.1145\/1240624.1240693"},{"key":"e_1_3_3_2_36_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544549.3583960"},{"key":"e_1_3_3_2_37_2","doi-asserted-by":"crossref","unstructured":"John Henderson. 2005. Introduction to real-world scene perception. Visual Cognition 12 6 (2005) 849\u2013851.","DOI":"10.1080\/13506280444000544"},{"key":"e_1_3_3_2_38_2","first-page":"1","volume-title":"Scene perception for psycholinguists","author":"Henderson J.M.","year":"2004","unstructured":"J.M. Henderson and Fernanda Ferreira. 2004. Scene perception for psycholinguists. Psychology Press, 1\u201358."},{"key":"e_1_3_3_2_39_2","doi-asserted-by":"publisher","DOI":"10.1145\/2207676.2208703"},{"key":"e_1_3_3_2_40_2","doi-asserted-by":"publisher","unstructured":"A.J. Hornof. 2004. Cognitive Strategies for the Visual Search of Hierarchical Computer Displays. Human\u2013Computer Interaction 19 3 (2004) 183\u2013223. 10.1207\/s15327051hci1903_1","DOI":"10.1207\/s15327051hci1903_1"},{"key":"e_1_3_3_2_41_2","doi-asserted-by":"crossref","unstructured":"Laurent Itti Christof Koch and Ernst Niebur. 1998. A model of saliency-based visual attention for rapid scene analysis. IEEE Transactions on pattern analysis and machine intelligence 20 11 (1998) 1254\u20131259.","DOI":"10.1109\/34.730558"},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613905.3638191"},{"key":"e_1_3_3_2_43_2","unstructured":"Yue Jiang. 2025. Computational representations for user interfaces. (2025)."},{"key":"e_1_3_3_2_44_2","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300643"},{"key":"e_1_3_3_2_45_2","doi-asserted-by":"publisher","DOI":"10.1145\/3654777.3676436"},{"key":"e_1_3_3_2_46_2","volume-title":"Workshop Paper at the 2023 CHI Conference on Human Factors in Computing Systems","author":"Jiang Yue","year":"2023","unstructured":"Yue Jiang, Luis\u00a0A Leiva, Hamed Rezazadegan\u00a0Tavakoli, Paul RB\u00a0Houssel, Julia Kylm\u00e4l\u00e4, and Antti Oulasvirta. 2023. UEyes: An Eye-Tracking Dataset across User Interface Types. In Workshop Paper at the 2023 CHI Conference on Human Factors in Computing Systems."},{"key":"e_1_3_3_2_47_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581096"},{"key":"e_1_3_3_2_48_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544549.3573805"},{"key":"e_1_3_3_2_49_2","doi-asserted-by":"publisher","DOI":"10.1145\/3491101.3504030"},{"key":"e_1_3_3_2_50_2","doi-asserted-by":"publisher","DOI":"10.1109\/VL\/HCC60511.2024.00032"},{"key":"e_1_3_3_2_51_2","doi-asserted-by":"publisher","DOI":"10.1145\/3708359.3712129"},{"key":"e_1_3_3_2_52_2","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445043"},{"key":"e_1_3_3_2_53_2","doi-asserted-by":"publisher","DOI":"10.1145\/3313831.3376610"},{"key":"e_1_3_3_2_54_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642822"},{"key":"e_1_3_3_2_55_2","doi-asserted-by":"publisher","DOI":"10.1145\/3025453.3025580"},{"key":"e_1_3_3_2_56_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2009.5459462"},{"key":"e_1_3_3_2_57_2","doi-asserted-by":"crossref","unstructured":"M.A. Just and P.A. Carpenter. 1976. Eye fixations and cognitive processes. Cognitive Psychology 8 4 (Oct. 1976) 441\u2013480.","DOI":"10.1016\/0010-0285(76)90015-3"},{"key":"e_1_3_3_2_58_2","doi-asserted-by":"publisher","DOI":"10.1145\/2556288.2557324"},{"key":"e_1_3_3_2_59_2","doi-asserted-by":"crossref","unstructured":"Matthias K\u00fcmmerer Matthias Bethge and Thomas\u00a0SA Wallis. 2022. DeepGaze III: Modeling free-viewing human scanpaths with deep learning. Journal of Vision 22 5 (2022).","DOI":"10.1167\/jov.22.5.7"},{"key":"e_1_3_3_2_60_2","doi-asserted-by":"crossref","unstructured":"Olivier Le\u00a0Meur Patrick Le\u00a0Callet and Dominique Barba. 2007. Predicting visual fixations on video based on low-level visual features. Vision research 47 19 (2007) 2483\u20132498.","DOI":"10.1016\/j.visres.2007.06.015"},{"key":"e_1_3_3_2_61_2","doi-asserted-by":"publisher","DOI":"10.1145\/3379503.3403557"},{"key":"e_1_3_3_2_62_2","unstructured":"Gang Li and Yang Li. 2022. Spotlight: Mobile UI Understanding using Vision-Language Models with a Focus. ArXiv abs\/2209.14927 (2022). https:\/\/api.semanticscholar.org\/CorpusID:252595735"},{"key":"e_1_3_3_2_63_2","first-page":"19730","volume-title":"International conference on machine learning","author":"Li Junnan","year":"2023","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven Hoi. 2023. Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In International conference on machine learning. PMLR, 19730\u201319742."},{"key":"e_1_3_3_2_64_2","first-page":"12888","volume-title":"International conference on machine learning","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven Hoi. 2022. Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In International conference on machine learning. PMLR, 12888\u201312900."},{"key":"e_1_3_3_2_65_2","unstructured":"Junnan Li Ramprasaath Selvaraju Akhilesh Gotmare Shafiq Joty Caiming Xiong and Steven Chu\u00a0Hong Hoi. 2021. Align before fuse: Vision and language representation learning with momentum distillation. Advances in neural information processing systems 34 (2021) 9694\u20139705."},{"key":"e_1_3_3_2_66_2","doi-asserted-by":"publisher","unstructured":"Jonathan Ling and Paul van Schaik. 2007. The influence of line spacing and text alignment on visual search of web pages. Displays 28 2 (2007) 60\u201367. 10.1016\/j.displa.2007.04.003","DOI":"10.1016\/j.displa.2007.04.003"},{"key":"e_1_3_3_2_67_2","unstructured":"Haotian Liu Chunyuan Li Yuheng Li Bo Li Yuanhan Zhang Sheng Shen and Yong\u00a0Jae Lee. 2024. LLaVA-NeXT: Improved reasoning OCR and world knowledge. https:\/\/llava-vl.github.io\/blog\/2024-01-30-llava-next\/"},{"key":"e_1_3_3_2_68_2","unstructured":"Haotian Liu Chunyuan Li Qingyang Wu and Yong\u00a0Jae Lee. 2023. Visual Instruction Tuning."},{"key":"e_1_3_3_2_69_2","doi-asserted-by":"publisher","unstructured":"Weilin Liu Yaqin Cao and R.W. Proctor. 2021. How do app icon color and border shape influence visual search efficiency and user experience? Evidence from an eye-tracking study. International Journal of Industrial Ergonomics 84 Article 103160 (2021). 10.1016\/j.ergon.2021.103160","DOI":"10.1016\/j.ergon.2021.103160"},{"key":"e_1_3_3_2_70_2","unstructured":"Yinhan Liu Myle Ott Naman Goyal Jingfei Du Mandar Joshi Danqi Chen Omer Levy Mike Lewis Luke Zettlemoyer and Veselin Stoyanov. 2019. Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1907.11692 (2019)."},{"key":"e_1_3_3_2_71_2","doi-asserted-by":"publisher","DOI":"10.1145\/3706599.3706736"},{"key":"e_1_3_3_2_72_2","unstructured":"Daniel Martin Diego Gutierrez and Belen Masia. 2022. A probabilistic time-evolving approach to scanpath prediction. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2204.09404 (2022)."},{"key":"e_1_3_3_2_73_2","doi-asserted-by":"crossref","unstructured":"Daniel Martin Ana Serrano Alexander\u00a0W Bergman Gordon Wetzstein and Belen Masia. 2022. Scangan360: A generative model of realistic scanpaths for 360 images. IEEE Transactions on Visualization and Computer Graphics 28 5 (2022) 2003\u20132013.","DOI":"10.1109\/TVCG.2022.3150502"},{"key":"e_1_3_3_2_74_2","doi-asserted-by":"publisher","DOI":"10.1145\/2702123.2702575"},{"key":"e_1_3_3_2_75_2","doi-asserted-by":"publisher","DOI":"10.1145\/375735.376414"},{"key":"e_1_3_3_2_76_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00145"},{"key":"e_1_3_3_2_77_2","doi-asserted-by":"crossref","unstructured":"Saul\u00a0B Needleman and Christian\u00a0D Wunsch. 1970. A general method applicable to the search for similarities in the amino acid sequence of two proteins. Journal of molecular biology 48 3 (1970) 443\u2013453.","DOI":"10.1016\/0022-2836(70)90057-4"},{"key":"e_1_3_3_2_78_2","doi-asserted-by":"publisher","unstructured":"M.B. Neider and G.J. Zelinsky. 2008. Exploring set size effects in scenes: Identifying the objects of search. Visual Cognition 16 1 (2008) 1\u201310. 10.1080\/13506280701381691","DOI":"10.1080\/13506280701381691"},{"key":"e_1_3_3_2_79_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-87702-8_8"},{"key":"e_1_3_3_2_80_2","first-page":"311","volume-title":"Proceedings of the 40th annual meeting of the Association for Computational Linguistics","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. Bleu: a method for automatic evaluation of machine translation. In Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311\u2013318."},{"key":"e_1_3_3_2_81_2","first-page":"466","volume-title":"European Conference on Computer Vision","author":"Peng Yi-Hao","year":"2024","unstructured":"Yi-Hao Peng, Faria Huq, Yue Jiang, Jason Wu, Xin\u00a0Yue Li, Jeffrey\u00a0P Bigham, and Amy Pavel. 2024. Dreamstruct: Understanding slides and user interfaces via synthetic data generation. In European Conference on Computer Vision. Springer, 466\u2013485."},{"key":"e_1_3_3_2_82_2","doi-asserted-by":"crossref","unstructured":"Robert\u00a0J Peters Asha Iyer Laurent Itti and Christof Koch. 2005. Components of bottom-up gaze allocation in natural images. Vision research 45 18 (2005) 2397\u20132416.","DOI":"10.1016\/j.visres.2005.03.019"},{"key":"e_1_3_3_2_83_2","doi-asserted-by":"publisher","DOI":"10.1145\/365024.365337"},{"key":"e_1_3_3_2_84_2","doi-asserted-by":"crossref","unstructured":"Aini Putkonen Yue Jiang Jingchun Zeng Olli Tammilehto Jussi\u00a0PP Jokinen and Antti Oulasvirta. 2025. Understanding visual search in graphical user interfaces. International Journal of Human-Computer Studies 199 (2025) 103483.","DOI":"10.1016\/j.ijhcs.2025.103483"},{"key":"e_1_3_3_2_85_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISMAR59233.2023.00037"},{"key":"e_1_3_3_2_86_2","unstructured":"Alec Radford Jong\u00a0Wook Kim Chris Hallacy Aditya Ramesh Gabriel Goh Sandhini Agarwal Girish Sastry Amanda Askell Pamela Mishkin Jack Clark Gretchen Krueger and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. arxiv:https:\/\/arXiv.org\/abs\/2103.00020\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2103.00020"},{"key":"e_1_3_3_2_87_2","unstructured":"Marc\u2019Aurelio Ranzato Sumit Chopra Michael Auli and Wojciech Zaremba. 2015. Sequence level training with recurrent neural networks. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1511.06732 (2015)."},{"key":"e_1_3_3_2_88_2","doi-asserted-by":"publisher","unstructured":"Hamed Rezazadegan Tavakoli Esa Rahtu and Janne Heikkil\u00e4. 2013. Stochastic bottom\u2013up fixation prediction and saccade generation. Image and Vision Computing 31 9 (2013) 686\u2013693. 10.1016\/j.imavis.2013.06.006","DOI":"10.1016\/j.imavis.2013.06.006"},{"key":"e_1_3_3_2_89_2","doi-asserted-by":"publisher","unstructured":"Ruth Rosenholtz Yuanzhen Li and Lisa Nakano. 2007. Measuring visual clutter. Journal of Vision 7 2 Article 17 (08 2007). 10.1167\/7.2.17","DOI":"10.1167\/7.2.17"},{"key":"e_1_3_3_2_90_2","doi-asserted-by":"crossref","unstructured":"Yossi Rubner Carlo Tomasi and Leonidas\u00a0J Guibas. 2000. The earth mover\u2019s distance as a metric for image retrieval. International journal of computer vision 40 2 (2000) 99\u2013121.","DOI":"10.1023\/A:1026543900054"},{"key":"e_1_3_3_2_91_2","unstructured":"John Schulman Filip Wolski Prafulla Dhariwal Alec Radford and Oleg Klimov. 2017. Proximal policy optimization algorithms. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1707.06347 (2017)."},{"key":"e_1_3_3_2_92_2","unstructured":"Zhihong Shao Peiyi Wang Qihao Zhu Runxin Xu Junxiao Song Xiao Bi Haowei Zhang Mingchuan Zhang YK Li Yang Wu et\u00a0al. 2024. Deepseekmath: Pushing the limits of mathematical reasoning in open language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2402.03300 (2024)."},{"key":"e_1_3_3_2_93_2","doi-asserted-by":"publisher","DOI":"10.1145\/2858036.2858469"},{"key":"e_1_3_3_2_94_2","first-page":"6989","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Sui Xiangjie","year":"2023","unstructured":"Xiangjie Sui, Yuming Fang, Hanwei Zhu, Shiqi Wang, and Zhou Wang. 2023. ScanDMM: A Deep Markov Model of Scanpath Prediction for 360deg Images. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 6989\u20136999."},{"key":"e_1_3_3_2_95_2","doi-asserted-by":"crossref","unstructured":"Wanjie Sun Zhenzhong Chen and Feng Wu. 2019. Visual scanpath prediction using IOR-ROI recurrent mixture density network. IEEE transactions on pattern analysis and machine intelligence 43 6 (2019) 2101\u20132118.","DOI":"10.1109\/TPAMI.2019.2956930"},{"key":"e_1_3_3_2_96_2","doi-asserted-by":"crossref","unstructured":"Michael\u00a0J Swain and Dana\u00a0H Ballard. 1991. Color indexing. International journal of computer vision 7 1 (1991) 11\u201332.","DOI":"10.1007\/BF00130487"},{"key":"e_1_3_3_2_97_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642777"},{"key":"e_1_3_3_2_98_2","unstructured":"Kimi Team Angang Du Bofei Gao Bowei Xing Changjiu Jiang Cheng Chen Cheng Li Chenjun Xiao Chenzhuang Du Chonghua Liao et\u00a0al. 2025. Kimi k1. 5: Scaling reinforcement learning with llms. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2501.12599 (2025)."},{"key":"e_1_3_3_2_99_2","doi-asserted-by":"publisher","DOI":"10.1145\/2207676.2208414"},{"key":"e_1_3_3_2_100_2","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300687"},{"key":"e_1_3_3_2_101_2","doi-asserted-by":"publisher","unstructured":"Kashyap Todi Jussi Jokinen Kris Luyten and Antti Oulasvirta. 2019. Individualising Graphical Layouts with Predictive Visual Search Models. ACM Transactions on Interactive Intelligent Systems 10 1 Article 9 (Aug. 2019) 24\u00a0pages. 10.1145\/3241381","DOI":"10.1145\/3241381"},{"key":"e_1_3_3_2_102_2","doi-asserted-by":"crossref","unstructured":"A.K. Trapp and Carolin Wienrich. 2018. App icon similarity and its impact on visual search efficiency on mobile touch devices. Cognitive Research: Principles and Implications 3 1 Article 39 (2018).","DOI":"10.1186\/s41235-018-0133-4"},{"key":"e_1_3_3_2_103_2","doi-asserted-by":"crossref","unstructured":"A.M. Treisman and Garry Gelade. 1980. A feature-integration theory of attention. Cognitive Psychology 12 1 (1980) 97\u2013136.","DOI":"10.1016\/0010-0285(80)90005-5"},{"key":"e_1_3_3_2_104_2","doi-asserted-by":"publisher","DOI":"10.1145\/1357054.1357221"},{"key":"e_1_3_3_2_105_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2011.5995423"},{"key":"e_1_3_3_2_106_2","unstructured":"Yao Wang Andreas Bulling et\u00a0al. 2023. Scanpath prediction on information visualisations. IEEE Transactions on Visualization and Computer Graphics (2023)."},{"key":"e_1_3_3_2_107_2","doi-asserted-by":"crossref","unstructured":"Yao Wang Yue Jiang Zhiming Hu Constantin Ruhdorfer Mihai B\u00e2ce and Andreas Bulling. 2024. VisRecall++: Analysing and predicting visualisation recallability from gaze behaviour. Proceedings of the ACM on Human-Computer Interaction 8 ETRA (2024) 1\u201318.","DOI":"10.1145\/3655613"},{"key":"e_1_3_3_2_108_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00336"},{"key":"e_1_3_3_2_109_2","doi-asserted-by":"crossref","unstructured":"J.M. Wolfe. 2021. Guided Search 6.0: An updated model of visual search. Psychonomic Bulletin & Review 28 4 (2021) 1060\u20131092.","DOI":"10.3758\/s13423-020-01859-9"},{"key":"e_1_3_3_2_110_2","doi-asserted-by":"crossref","unstructured":"J.M. Wolfe and T.S. Horowitz. 2004. What attributes guide the deployment of visual attention and how do they do it? Nature Reviews Neuroscience 5 6 (2004) 495\u2013501.","DOI":"10.1038\/nrn1411"},{"key":"e_1_3_3_2_111_2","doi-asserted-by":"publisher","unstructured":"J.M. Wolfe E.M. Palmer and T.S. Horowitz. 2010. Reaction time distributions constrain models of visual search. Vision Research 50 14 (2010) 1304\u20131311. 10.1016\/j.visres.2009.11.002","DOI":"10.1016\/j.visres.2009.11.002"},{"key":"e_1_3_3_2_112_2","doi-asserted-by":"publisher","unstructured":"Chen Xia Junwei Han Fei Qi and Guangming Shi. 2019. Predicting Human Saccadic Scanpaths Based on Iterative Representation Learning. IEEE Transactions on Image Processing 28 7 (2019) 3502\u20133515. 10.1109\/TIP.2019.2897966","DOI":"10.1109\/TIP.2019.2897966"},{"key":"e_1_3_3_2_113_2","doi-asserted-by":"crossref","unstructured":"Mai Xu Yuhang Song Jianyi Wang MingLang Qiao Liangyu Huo and Zulin Wang. 2018. Predicting head movement in panoramic video: A deep reinforcement learning approach. IEEE transactions on pattern analysis and machine intelligence 41 11 (2018) 2693\u20132708.","DOI":"10.1109\/TPAMI.2018.2858783"},{"key":"e_1_3_3_2_114_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00027"},{"key":"e_1_3_3_2_115_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.503"},{"key":"e_1_3_3_2_116_2","doi-asserted-by":"publisher","DOI":"10.1145\/3313831.3376870"},{"key":"e_1_3_3_2_117_2","doi-asserted-by":"publisher","unstructured":"G.J. Zelinsky. 1996. Using Eye Saccades to Assess the Selectivity of Search Movements. Vision Research 36 14 (1996) 2177\u20132187. 10.1016\/0042-6989(95)00300-2","DOI":"10.1016\/0042-6989(95)00300-2"}],"event":{"name":"CHI 2026: CHI Conference on Human Factors in Computing Systems","location":"Barcelona Spain","acronym":"CHI '26","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 2026 CHI Conference on Human Factors in Computing Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3772318.3791178","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,17]],"date-time":"2026-04-17T09:49:47Z","timestamp":1776419387000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3772318.3791178"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,13]]},"references-count":116,"alternative-id":["10.1145\/3772318.3791178","10.1145\/3772318"],"URL":"https:\/\/doi.org\/10.1145\/3772318.3791178","relation":{},"subject":[],"published":{"date-parts":[[2026,4,13]]},"assertion":[{"value":"2026-04-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}