{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,21]],"date-time":"2026-06-21T01:49:37Z","timestamp":1782006577208,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":38,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,5,31]],"date-time":"2026-05-31T00:00:00Z","timestamp":1780185600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/100000002","name":"NIH (National Institutes of Health)","doi-asserted-by":"publisher","award":["R01-HL171376"],"award-info":[{"award-number":["R01-HL171376"]}],"id":[{"id":"10.13039\/100000002","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000002","name":"NIH (National Institutes of Health)","doi-asserted-by":"publisher","award":["U01-CA268808"],"award-info":[{"award-number":["U01-CA268808"]}],"id":[{"id":"10.13039\/100000002","id-type":"DOI","asserted-by":"publisher"}]},{"name":"F-Initiatives","award":["Compute"],"award-info":[{"award-number":["Compute"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1145\/3797246.3806223","type":"proceedings-article","created":{"date-parts":[[2026,5,29]],"date-time":"2026-05-29T12:08:10Z","timestamp":1780056490000},"page":"1-7","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["What They Saw, Not Just Where They Looked: Semantic Scanpath Similarity via VLMs and NLP metrics"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7479-6879","authenticated-orcid":false,"given":"Mohamed Amine","family":"Kerkouri","sequence":"first","affiliation":[{"name":"f-initiatives, Paris, France"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1696-7764","authenticated-orcid":false,"given":"Marouane","family":"Tliba","sequence":"additional","affiliation":[{"name":"PRISME, University of Orleans, Orleans, France"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8655-1134","authenticated-orcid":false,"given":"Bin","family":"Wang","sequence":"additional","affiliation":[{"name":"Northwestern University, Evanston, Illinois, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2066-4707","authenticated-orcid":false,"given":"Aladine","family":"Chetouani","sequence":"additional","affiliation":[{"name":"Universit\u00e9 Sorbonne Paris Nord, Villetaneuse, France"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7379-6829","authenticated-orcid":false,"given":"Ulas","family":"Bagci","sequence":"additional","affiliation":[{"name":"Radiology, Northwestern University, Chicago, Illinois, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0707-6131","authenticated-orcid":false,"given":"Alessandro","family":"Bruno","sequence":"additional","affiliation":[{"name":"IULM AI Lab, Milan, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,5,31]]},"reference":[{"key":"e_1_3_3_2_2_1","doi-asserted-by":"crossref","unstructured":"Momina\u00a0Liaqat Ali and Zhou Zhang. 2024. The YOLO framework: A comprehensive review of evolution applications and benchmarks in object detection. Computers 13 12 (2024) 336.","DOI":"10.3390\/computers13120336"},{"key":"e_1_3_3_2_3_1","unstructured":"Jinze Bai Shuai Bai Shusheng Yang Shijie Wang Sinan Tan Peng Wang Junyang Lin Chang Zhou and Jingren Zhou. 2023. Qwen-VL: A Versatile Vision-Language Model for Understanding Localization Text Reading and Beyond. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2308.12966 (2023)."},{"key":"e_1_3_3_2_4_1","unstructured":"Shuai Bai Yuxuan Cai Ruizhe Chen Keqin Chen Xionghui Chen Zesen Cheng Lianghao Deng Wei Ding Chang Gao Chunjiang Ge Wenbin Ge Zhifang Guo Qidong Huang Jie Huang Fei Huang Binyuan Hui Shutong Jiang Zhaohai Li Mingsheng Li Mei Li Kaixin Li Zicheng Lin Junyang Lin Xuejing Liu Jiawei Liu Chenglong Liu Yang Liu Dayiheng Liu Shixuan Liu Dunjie Lu Ruilin Luo Chenxu Lv Rui Men Lingchen Meng Xuancheng Ren Xingzhang Ren Sibo Song Yuchong Sun Jun Tang Jianhong Tu Jianqiang Wan Peng Wang Pengfei Wang Qiuyue Wang Yuxuan Wang Tianbao Xie Yiheng Xu Haiyang Xu Jin Xu Zhibo Yang Mingkun Yang Jianxin Yang An Yang Bowen Yu Fei Zhang Hang Zhang Xi Zhang Bo Zheng Humen Zhong Jingren Zhou Fan Zhou Jing Zhou Yuanzhi Zhu and Ke Zhu. 2025. Qwen3-VL Technical Report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2511.21631 (2025)."},{"key":"e_1_3_3_2_5_1","doi-asserted-by":"publisher","DOI":"10.5555\/3000850.3000887"},{"key":"e_1_3_3_2_6_1","unstructured":"Florian Bordes Richard\u00a0Yuanzhe Pang Anurag Ajay Alexander\u00a0C Li Adrien Bardes Suzanne Petryk Oscar Ma\u00f1as Zhiqiu Lin Anas Mahmoud Bargav Jayaraman et\u00a0al. 2024. An introduction to vision-language modeling. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2405.17247 (2024)."},{"key":"e_1_3_3_2_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3588015.3590133"},{"key":"e_1_3_3_2_8_1","doi-asserted-by":"crossref","unstructured":"\u00dcmit\u00a0Can B\u00fcy\u00fckakg\u00fcl Arif Y\u00fcce and Hakan Kat\u0131rc\u0131. 2025. Where Vision Meets Memory: An Eye-Tracking Study of In-App Ads in Mobile Sports Games with Mixed Visual-Quantitative Analytics. Journal of Eye Movement Research 18 6 (2025) 74.","DOI":"10.3390\/jemr18060074"},{"key":"e_1_3_3_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW56347.2022.00551"},{"key":"e_1_3_3_2_10_1","doi-asserted-by":"crossref","unstructured":"Filipe Cristino Sebastiaan Math\u00f4t Jan Theeuwes and Iain\u00a0D Gilchrist. 2010. ScanMatch: A novel method for comparing fixation sequences. Behavior research methods 42 3 (2010) 692\u2013700.","DOI":"10.3758\/BRM.42.3.692"},{"key":"e_1_3_3_2_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/1743666.1743719"},{"key":"e_1_3_3_2_12_1","unstructured":"Yongchao Feng Yajie Liu Shuai Yang Wenrui Cai Jinqing Zhang Qiqi Zhan Ziyue Huang Hongxi Yan Qiao Wan Chenguang Liu Junzhe Wang Jiahui Lv Ziqi Liu Teng Shi Qingjie Liu and Yunhong Wang. 2025. Vision-Language Model for Object Detection and Segmentation: A Review and Evaluation. ArXiv abs\/2504.09480 (2025). https:\/\/api.semanticscholar.org\/CorpusID:277781245"},{"key":"e_1_3_3_2_13_1","doi-asserted-by":"crossref","unstructured":"He Huang Philipp Doebler and Barbara Mertins. 2024. Short-time AOIs-based representative scanpath identification and scanpath aggregation. Behavior Research Methods 56 6 (2024) 6051\u20136066.","DOI":"10.3758\/s13428-023-02332-w"},{"key":"e_1_3_3_2_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/1743666.1743718"},{"key":"e_1_3_3_2_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581096"},{"key":"e_1_3_3_2_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3549555.3549597"},{"key":"e_1_3_3_2_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3649902.3655656"},{"key":"e_1_3_3_2_18_1","unstructured":"Mohamed\u00a0Amine Kerkouri Marouane Tliba Aladine Chetouani and Alessandro Bruno. 2026. SPGen: Stochastic scanpath generation for paintings using unsupervised domain adaptation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2602.22049 (2026)."},{"key":"e_1_3_3_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP42928.2021.9506295"},{"key":"e_1_3_3_2_20_1","doi-asserted-by":"publisher","unstructured":"Mohamed\u00a0Amine Kerkouri Marouane Tliba Aladine Chetouani and Mohamed Sayeh. 2022b. SalyPath360: Saliency and scanpath prediction framework for omnidirectional images. Electronic Imaging 34 11 (2022) 168\u20131\u2013168\u20131. 10.2352\/EI.2022.34.11.HVEI-168","DOI":"10.2352\/EI.2022.34.11.HVEI-168"},{"key":"e_1_3_3_2_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_3_2_22_1","first-page":"707","volume-title":"Soviet physics doklady","author":"Levenshtein Vladimir\u00a0I","year":"1966","unstructured":"Vladimir\u00a0I Levenshtein et\u00a0al. 1966. Binary codes capable of correcting deletions, insertions, and reversals. In Soviet physics doklady , Vol.\u00a010. Soviet Union, 707\u2013710."},{"key":"e_1_3_3_2_23_1","first-page":"74","volume-title":"Text summarization branches out","author":"Lin Chin-Yew","year":"2004","unstructured":"Chin-Yew Lin. 2004. Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74\u201381."},{"key":"e_1_3_3_2_24_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"e_1_3_3_2_25_1","unstructured":"Haotian Liu Chunyuan Li Qingyang Wu and Yong\u00a0Jae Lee. 2023. Visual Instruction Tuning."},{"key":"e_1_3_3_2_26_1","doi-asserted-by":"crossref","unstructured":"Abdulrahman Mohamed\u00a0Selim Michael Barz Omair\u00a0Shahzad Bhatti Hasan Md\u00a0Tusfiqur Alam and Daniel Sonntag. 2024. A review of machine learning in scanpath analysis for passive gaze-based interaction. Frontiers in Artificial Intelligence 7 (2024) 1391745.","DOI":"10.3389\/frai.2024.1391745"},{"key":"e_1_3_3_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.00263"},{"key":"e_1_3_3_2_28_1","first-page":"311","volume-title":"Proceedings of the 40th annual meeting of the Association for Computational Linguistics","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. Bleu: a method for automatic evaluation of machine translation. In Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311\u2013318."},{"key":"e_1_3_3_2_29_1","first-page":"8748","volume-title":"International conference on machine learning","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et\u00a0al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PmLR, 8748\u20138763."},{"key":"e_1_3_3_2_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/2578153.2578173"},{"key":"e_1_3_3_2_31_1","series-title":"(NIPS\u201915)","first-page":"91","volume-title":"Proceedings of the 29th International Conference on Neural Information Processing Systems - Volume 1","author":"Ren Shaoqing","year":"2015","unstructured":"Shaoqing Ren, Kaiming He, Ross Girshick, and Jian Sun. 2015. Faster R-CNN: towards real-time object detection with region proposal networks. In Proceedings of the 29th International Conference on Neural Information Processing Systems - Volume 1 (Montreal, Canada) (NIPS\u201915). MIT Press, Cambridge, MA, USA, 91\u201399."},{"key":"e_1_3_3_2_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3729905"},{"key":"e_1_3_3_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW56347.2022.00160"},{"key":"e_1_3_3_2_34_1","doi-asserted-by":"crossref","unstructured":"Marouane Tliba Mohamed\u00a0A Kerkouri Bashir Ghariba Aladine Chetouani Arzu \u00c7\u00f6ltekin Mohamed\u00a0Sami Shehata and Alessandro Bruno. 2022b. Satsal: A multi-level self-attention based architecture for visual saliency prediction. IEEE Access 10 (2022) 20701\u201320713.","DOI":"10.1109\/ACCESS.2022.3152189"},{"key":"e_1_3_3_2_35_1","doi-asserted-by":"crossref","unstructured":"Alexander\u00a0JA Ty Zheng Fang Rivver\u00a0A Gonzalez Paul\u00a0J Rozdeba and Henry\u00a0DI Abarbanel. 2019. Machine learning of time series using time-delay embedding and precision annealing. Neural Computation 31 10 (2019) 2004\u20132024.","DOI":"10.1162\/neco_a_01224"},{"key":"e_1_3_3_2_36_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i8.32881"},{"key":"e_1_3_3_2_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3715669.3726789"},{"key":"e_1_3_3_2_38_1","unstructured":"Zhibo Yang Sounak Mondal Seoyoung Ahn Gregory Zelinsky Minh Hoai and Dimitris Samaras. 2023. Predicting Human Attention using Computational Attention. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.09383 (2023)."},{"key":"e_1_3_3_2_39_1","unstructured":"Tianyi Zhang Varsha Kishore Felix Wu Kilian\u00a0Q Weinberger and Yoav Artzi. 2019. Bertscore: Evaluating text generation with bert. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1904.09675 (2019)."}],"event":{"name":"ETRA '26: 2026 Symposium on Eye Tracking Research and Applications","location":"Marrakesh Morocco","acronym":"ETRA '26","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction","SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Proceedings of the 2026 Symposium on Eye Tracking Research and Applications"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3797246.3806223","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,21]],"date-time":"2026-06-21T01:13:53Z","timestamp":1782004433000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3797246.3806223"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,31]]},"references-count":38,"alternative-id":["10.1145\/3797246.3806223","10.1145\/3797246"],"URL":"https:\/\/doi.org\/10.1145\/3797246.3806223","relation":{},"subject":[],"published":{"date-parts":[[2026,5,31]]},"assertion":[{"value":"2026-05-31","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}