{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T17:24:55Z","timestamp":1777656295983,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":67,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100021856","name":"Ministero dell'Universit\u00e0 e della Ricerca","doi-asserted-by":"publisher","award":["PE00000013"],"award-info":[{"award-number":["PE00000013"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100021856","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100021856","name":"Ministero dell'Universit\u00e0 e della Ricerca","doi-asserted-by":"publisher","award":["2020TA3K9N"],"award-info":[{"award-number":["2020TA3K9N"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100021856","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/100010661","name":"Horizon 2020 Framework Programme","doi-asserted-by":"publisher","award":["871245"],"award-info":[{"award-number":["871245"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/100010661","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/100018693","name":"HORIZON EUROPE Framework Programme","doi-asserted-by":"publisher","award":["101120237"],"award-info":[{"award-number":["101120237"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/100018693","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3680952","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:41Z","timestamp":1729925981000},"page":"2360-2369","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["AL-GTD: Deep Active Learning for Gaze Target Detection"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1938-3449","authenticated-orcid":false,"given":"Francesco","family":"Tonini","sequence":"first","affiliation":[{"name":"University of Trento &amp; Fondazione Bruno Kessler, Trento, Italy"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-6781-5770","authenticated-orcid":false,"given":"Nicola","family":"Dall'Asen","sequence":"additional","affiliation":[{"name":"University of Trento &amp; University of Pisa, Trento, Italy"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1874-3078","authenticated-orcid":false,"given":"Lorenzo","family":"Vaquero","sequence":"additional","affiliation":[{"name":"Fondazione Bruno Kessler, Trento, Italy"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9583-0087","authenticated-orcid":false,"given":"Cigdem","family":"Beyan","sequence":"additional","affiliation":[{"name":"Department of Computer Science, University of Verona, Verona, Italy"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0228-1147","authenticated-orcid":false,"given":"Elisa","family":"Ricci","sequence":"additional","affiliation":[{"name":"University of Trento &amp; Fondazione Bruno Kessler, Trento, Italy"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.5898\/JHRI.6.1.Admoni"},{"key":"e_1_3_2_2_2_1","volume-title":"Proc. of the IEEE\/CVF International Conference on Computer Vision. IEEE","author":"Aghdam Hamed H","unstructured":"Hamed H Aghdam, Abel Gonzalez-Garcia, Joost van de Weijer, and Antonio M L\u00f3pez. 2019. Active learning for deep detection neural networks. In Proc. of the IEEE\/CVF International Conference on Computer Vision. IEEE, Seoul, Korea (South), 3672--3680."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.16910\/jemr.13.4.5"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01373"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00976"},{"key":"e_1_3_2_2_6_1","volume-title":"Comput. Surveys","volume":"56","author":"Beyan Cigdem","year":"2023","unstructured":"Cigdem Beyan, Alessandro Vinciarelli, and Alessio Del Bue. 2023. Co-Located Human-Human Interaction Analysis using Nonverbal Cues: A Survey. Comput. Surveys, Vol. 56, 5 (2023), 109:1--109:41."},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01084"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02110"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.isci.2019.05.035"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00946"},{"key":"e_1_3_2_2_11_1","volume-title":"Proc","author":"Cheng Yihua","unstructured":"Yihua Cheng, Feng Lu, and Xucong Zhang. 2018. Appearance-Based Gaze Estimation via Evaluation-Guided Asymmetric Regression. In Proc. of ECCV. Springer, Munich, Germany, 105--121."},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2021.104471"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01010"},{"key":"e_1_3_2_2_14_1","volume-title":"The European Conference on Computer Vision (ECCV). Springer","author":"Chong Eunji","unstructured":"Eunji Chong, Nataniel Ruiz, Yongxin Wang, Yun Zhang, Agata Rozga, and James M. Rehg. 2018. Connecting Gaze, Scene, and Attention: Generalized Attention Estimation via Joint Modeling of Gaze and Scene Saliency. In The European Conference on Computer Vision (ECCV). Springer, Munich, Germany,, 397--412."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00544"},{"key":"e_1_3_2_2_16_1","volume-title":"Hill H Goldsmith, Andrew L Alexander, and Richard J Davidson.","author":"Dalton Kim M","year":"2005","unstructured":"Kim M Dalton, Brendon M Nacewicz, Tom Johnstone, Hillary S Schaefer, Morton Ann Gernsbacher, Hill H Goldsmith, Andrew L Alexander, and Richard J Davidson. 2005. Gaze fixation and the neural circuitry of face processing in autism. Nature neuroscience, Vol. 8, 4 (2005), 519--526."},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3204949.3208139"},{"key":"e_1_3_2_2_18_1","volume-title":"Adversarial active learning for deep networks: a margin based approach. CoRR","author":"Ducoffe Melanie","year":"2018","unstructured":"Melanie Ducoffe and Frederic Precioso. 2018. Adversarial active learning for deep networks: a margin based approach. CoRR, Vol. abs\/1802.09841 (2018), 1--10."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1098\/rspb.2015.1141"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01409"},{"key":"e_1_3_2_2_21_1","volume-title":"The eyes have it: the neuroethology, function and evolution of social gaze. Neuroscience & biobehavioral reviews","author":"Emery Nathan J","year":"2000","unstructured":"Nathan J Emery. 2000. The eyes have it: the neuroethology, function and evolution of social gaze. Neuroscience & biobehavioral reviews, Vol. 24, 6 (2000), 581--604."},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1177\/25152459221147250"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01123"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/2578153.2578190"},{"key":"e_1_3_2_2_25_1","volume-title":"international conference on machine learning. PMLR, JMLR.org","author":"Gal Yarin","year":"2016","unstructured":"Yarin Gal and Zoubin Ghahramani. 2016. Dropout as a bayesian approximation: Representing model uncertainty in deep learning. In international conference on machine learning. PMLR, JMLR.org, New York City, NY, USA, 1050--1059."},{"key":"e_1_3_2_2_26_1","volume-title":"International conference on machine learning. PMLR, PMLR","author":"Gal Yarin","year":"2017","unstructured":"Yarin Gal, Riashat Islam, and Zoubin Ghahramani. 2017. Deep bayesian active learning with image data. In International conference on machine learning. PMLR, PMLR, Sydney, NSW, Australia, 1183--1192."},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP.2016.7532697"},{"key":"e_1_3_2_2_28_1","volume-title":"Deep active learning over the long tail. CoRR","author":"Geifman Yonatan","year":"2017","unstructured":"Yonatan Geifman and Ran El-Yaniv. 2017. Deep active learning over the long tail. CoRR, Vol. abs\/1711.00941 (2017), 1--10."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01080"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/TBME.2018.2889915"},{"key":"e_1_3_2_2_31_1","volume-title":"Proc. of the Asian Conference on Computer Vision. Springer","author":"Guo Zidong","year":"2020","unstructured":"Zidong Guo, Zejian Yuan, Chong Zhang, Wanchao Chi, Yonggen Ling, and Shenghao Zhang. 2020. Domain adaptation gaze estimation by embedding with prediction consistency. In Proc. of the Asian Conference on Computer Vision. Springer, Kyoto, Japan, 292--307."},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01252-6_13"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIM.2022.3160534"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2022.104924"},{"key":"e_1_3_2_2_35_1","volume-title":"Computer Vision--ACCV 2018: 14th Asian Conference on Computer Vision","author":"Kao Chieh-Chi","unstructured":"Chieh-Chi Kao, Teng-Yok Lee, Pradeep Sen, and Ming-Yu Liu. 2019. Localization-aware active learning for object detection. In Computer Vision--ACCV 2018: 14th Asian Conference on Computer Vision. Springer, Springer, Perth, Australia, 506--522."},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00701"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00807"},{"key":"e_1_3_2_2_38_1","volume-title":"Advances in Neural Information Processing Systems. NeurIPS","author":"Kirsch Andreas","unstructured":"Andreas Kirsch, Joost van Amersfoort, and Yarin Gal. 2019. BatchBALD: Efficient and Diverse Batch Acquisition for Deep Bayesian Active Learning. In Advances in Neural Information Processing Systems. NeurIPS, Vancouver, BC, Canada, 7024--7035."},{"key":"e_1_3_2_2_39_1","volume-title":"Advances in Neural Information Processing Systems","author":"Lakshminarayanan Balaji","unstructured":"Balaji Lakshminarayanan, Alexander Pritzel, and Charles Blundell. 2017. Simple and scalable predictive uncertainty estimation using deep ensembles. In Advances in Neural Information Processing Systems. NIPS, Long Beach, CA, USA, 6402--6413."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3051319"},{"key":"e_1_3_2_2_41_1","volume-title":"Asian Conference on Computer Vision. Springer","author":"Lian Dongze","year":"2018","unstructured":"Dongze Lian, Zehao Yu, and Shenghua Gao. 2018. Believe it or not, we know what you are looking at!. In Asian Conference on Computer Vision. Springer, Springer, Perth, Australia, 35--50."},{"key":"e_1_3_2_2_42_1","volume-title":"Computer Vision--ECCV 2014: 13th European Conference","author":"Lin Tsung-Yi","unstructured":"Tsung-Yi Lin, Michael Maire, Serge Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Doll\u00e1r, and C Lawrence Zitnick. 2014. Microsoft coco: Common objects in context. In Computer Vision--ECCV 2014: 13th European Conference. Springer, Springer, Zurich, Switzerland, 740--755."},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00622"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00539"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02276"},{"key":"e_1_3_2_2_46_1","volume-title":"Image Analysis and Processing--ICIAP 2022: 21st International Conference","author":"Mazzamuto Michele","unstructured":"Michele Mazzamuto, Francesco Ragusa, Antonino Furnari, Giovanni Signorello, and Giovanni Maria Farinella. 2022. Weakly Supervised Attended Object Detection Using Gaze Data as Annotations. In Image Analysis and Processing--ICIAP 2022: 21st International Conference. Springer, Lecce, Italy, 263--274."},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV56688.2023.00094"},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00032"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.5555\/1061935.1649091"},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01809"},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2020.3019967"},{"key":"e_1_3_2_2_52_1","volume-title":"Eye movements in reading and information processing: 20 years of research. Psychological bulletin","author":"Rayner Keith","year":"1998","unstructured":"Keith Rayner. 1998. Eye movements in reading and information processing: 20 years of research. Psychological bulletin, Vol. 124, 3 (1998), 372."},{"key":"e_1_3_2_2_53_1","volume-title":"Advances in Neural Information Processing Systems","volume":"28","author":"Recasens Adria","year":"2015","unstructured":"Adria Recasens, Aditya Khosla, Carl Vondrick, and Antonio Torralba. 2015. Where are they looking?. In Advances in Neural Information Processing Systems, Vol. 28. Curran Associates, Inc., Montreal, Quebec, Canada, 199--207."},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.160"},{"key":"e_1_3_2_2_55_1","volume-title":"British Machine Vision Conference","volume":"362","author":"Roy Soumya","year":"2018","unstructured":"Soumya Roy, Asim Unmesh, and Vinay P Namboodiri. 2018. Deep active learning for object detection.. In British Machine Vision Conference, Vol. 362. BMVA Press, Newcastle, UK, 91."},{"key":"e_1_3_2_2_56_1","volume-title":"Active Learning for Convolutional Neural Networks: A Core-Set Approach. In International Conference on Learning Representations. OpenReview.net","author":"Sener Ozan","year":"2018","unstructured":"Ozan Sener and Silvio Savarese. 2018. Active Learning for Convolutional Neural Networks: A Core-Set Approach. In International Conference on Learning Representations. OpenReview.net, Vancouver, BC, Canada, 1--13."},{"key":"e_1_3_2_2_57_1","volume-title":"A mathematical theory of communication. ACM SIGMOBILE mobile computing and communications review","author":"Shannon Claude Elwood","year":"2001","unstructured":"Claude Elwood Shannon. 2001. A mathematical theory of communication. ACM SIGMOBILE mobile computing and communications review, Vol. 5, 1 (2001), 3--55."},{"key":"e_1_3_2_2_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00607"},{"key":"e_1_3_2_2_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/3536221.3556624"},{"key":"e_1_3_2_2_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01998"},{"key":"e_1_3_2_2_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00224"},{"key":"e_1_3_2_2_62_1","volume-title":"Deep active learning for video-based person re-identification. CoRR","author":"Wang Menglin","year":"2018","unstructured":"Menglin Wang, Baisheng Lai, Zhongming Jin, Xiaojin Gong, Jianqiang Huang, and Xiansheng Hua. 2018. Deep active learning for video-based person re-identification. CoRR, Vol. abs\/1812.05785 (2018), 1--9."},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00918"},{"key":"e_1_3_2_2_64_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00790"},{"key":"e_1_3_2_2_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00018"},{"key":"e_1_3_2_2_66_1","volume-title":"Advances in Neural Information Processing Systems","volume":"27","author":"Zhou Bolei","year":"2014","unstructured":"Bolei Zhou, Agata Lapedriza, Jianxiong Xiao, Antonio Torralba, and Aude Oliva. 2014. Learning deep features for scene recognition using places database. In Advances in Neural Information Processing Systems, Vol. 27. NIPS, Montreal, Quebec, 487--495."},{"key":"e_1_3_2_2_67_1","volume-title":"Deformable DETR: Deformable Transformers for End-to-End Object Detection. In 9th International Conference on Learning Representations, ICLR","author":"Zhu Xizhou","year":"2021","unstructured":"Xizhou Zhu, Weijie Su, Lewei Lu, Bin Li, Xiaogang Wang, and Jifeng Dai. 2021. Deformable DETR: Deformable Transformers for End-to-End Object Detection. In 9th International Conference on Learning Representations, ICLR 2021. OpenReview.net, Virtual, 1--16."}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3680952","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3680952","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:34Z","timestamp":1750295854000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3680952"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":67,"alternative-id":["10.1145\/3664647.3680952","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3680952","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}