{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T23:35:23Z","timestamp":1783035323353,"version":"3.54.6"},"publisher-location":"Cham","reference-count":50,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031728891","type":"print"},{"value":"9783031728907","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,12,7]],"date-time":"2024-12-07T00:00:00Z","timestamp":1733529600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,7]],"date-time":"2024-12-07T00:00:00Z","timestamp":1733529600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72890-7_23","type":"book-chapter","created":{"date-parts":[[2024,12,6]],"date-time":"2024-12-06T20:02:38Z","timestamp":1733515358000},"page":"369-386","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Boosting Gaze Object Prediction via\u00a0Pixel-Level Supervision from\u00a0Vision Foundation Model"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-3285-427X","authenticated-orcid":false,"given":"Yang","family":"Jin","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1520-9360","authenticated-orcid":false,"given":"Lei","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6123-9870","authenticated-orcid":false,"given":"Shi","family":"Yan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8028-0166","authenticated-orcid":false,"given":"Bin","family":"Fan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9266-4685","authenticated-orcid":false,"given":"Binglu","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,12,7]]},"reference":[{"key":"23_CR1","doi-asserted-by":"crossref","unstructured":"Bao, J., Liu, B., Yu, J.: ESCNet: gaze target detection with the understanding of 3D scenes. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 14126\u201314135 (2022)","DOI":"10.1109\/CVPR52688.2022.01373"},{"key":"23_CR2","doi-asserted-by":"crossref","unstructured":"Cai, X., Zeng, J., Shan, S., Chen, X.: Source-free adaptive gaze estimation by uncertainty reduction. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 22035\u201322045 (2023)","DOI":"10.1109\/CVPR52729.2023.02110"},{"key":"23_CR3","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"213","DOI":"10.1007\/978-3-030-58452-8_13","volume-title":"Computer Vision \u2013 ECCV 2020","author":"N Carion","year":"2020","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12346, pp. 213\u2013229. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58452-8_13"},{"issue":"2","key":"23_CR4","doi-asserted-by":"publisher","first-page":"493","DOI":"10.1038\/s41591-022-02180-9","volume":"29","author":"W Chen","year":"2023","unstructured":"Chen, W., et al.: Early detection of visual impairment in young children using a smartphone-based deep learning system. Nat. Med. 29(2), 493\u2013503 (2023)","journal-title":"Nat. Med."},{"key":"23_CR5","doi-asserted-by":"crossref","unstructured":"Chen, Y., Nan, Z., Xiang, T.: FBLNet: feedback loop network for driver attention prediction. In: International Conference on Computer Vision, pp. 13371\u201313380 (2023)","DOI":"10.1109\/ICCV51070.2023.01230"},{"key":"23_CR6","doi-asserted-by":"crossref","unstructured":"Cheng, Y., Lu, F.: Gaze estimation using transformer. In: International Conference on Pattern Recognition, pp. 3341\u20133347 (2022)","DOI":"10.1109\/ICPR56361.2022.9956687"},{"key":"23_CR7","doi-asserted-by":"crossref","unstructured":"Cheng, Y., Lu, F., Zhang, X.: Appearance-based gaze estimation via evaluation-guided asymmetric regression. In: European Conference on Computer Vision, pp. 100\u2013115 (2018)","DOI":"10.1007\/978-3-030-01264-9_7"},{"key":"23_CR8","doi-asserted-by":"crossref","unstructured":"Chong, E., Ruiz, N., Wang, Y., Zhang, Y., Rozga, A., Rehg, J.M.: Connecting gaze, scene, and attention: generalized attention estimation via joint modeling of gaze and scene saliency. In: European Conference on Computer Vision, pp. 383\u2013398 (2018)","DOI":"10.1007\/978-3-030-01228-1_24"},{"key":"23_CR9","doi-asserted-by":"crossref","unstructured":"Chong, E., Wang, Y., Ruiz, N., Rehg, J.M.: Detecting attended visual targets in video. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 5396\u20135406 (2020)","DOI":"10.1109\/CVPR42600.2020.00544"},{"key":"23_CR10","doi-asserted-by":"crossref","unstructured":"Gupta, A., Tafasca, S., Odobez, J.M.: A modular multimodal architecture for gaze target prediction: application to privacy-sensitive settings. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 5041\u20135050 (2022)","DOI":"10.1109\/CVPRW56347.2022.00552"},{"key":"23_CR11","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., Girshick, R.: Mask R-CNN. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 2961\u20132969 (2017)","DOI":"10.1109\/ICCV.2017.322"},{"key":"23_CR12","first-page":"1","volume":"71","author":"Z Hu","year":"2022","unstructured":"Hu, Z., Yang, D., Cheng, S., Zhou, L., Wu, S., Liu, J.: We know where they are looking at from the RGB-D camera: gaze following in 3D. IEEE Trans. Instrum. Meas. 71, 1\u201314 (2022)","journal-title":"IEEE Trans. Instrum. Meas."},{"key":"23_CR13","doi-asserted-by":"crossref","unstructured":"Hu, Z., Yang, Y., Zhai, X., Yang, D., Zhou, B., Liu, J.: GFIE: a dataset and baseline for gaze-following from 2D to 3D in indoor environments. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 8907\u20138916 (2023)","DOI":"10.1109\/CVPR52729.2023.00860"},{"issue":"12","key":"23_CR14","doi-asserted-by":"publisher","first-page":"8524","DOI":"10.1109\/TCSVT.2022.3190314","volume":"32","author":"Z Hu","year":"2022","unstructured":"Hu, Z., et al.: Gaze target estimation inspired by interactive attention. IEEE Trans. Circuit Syst. Video Technol. 32(12), 8524\u20138536 (2022)","journal-title":"IEEE Trans. Circuit Syst. Video Technol."},{"issue":"10","key":"23_CR15","doi-asserted-by":"publisher","first-page":"19374","DOI":"10.1109\/TITS.2022.3166208","volume":"23","author":"T Huang","year":"2022","unstructured":"Huang, T., Fu, R.: Driver distraction detection based on the true driver\u2019s focus of attention. IEEE Trans. Intell. Transp. Syst. 23(10), 19374\u201319386 (2022)","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"23_CR16","doi-asserted-by":"crossref","unstructured":"Jin, S., Wang, Z., Wang, L., Bi, N., Nguyen, T.: ReDirTrans: latent-to-latent translation for gaze and head redirection. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 5547\u20135556 (2023)","DOI":"10.1109\/CVPR52729.2023.00537"},{"key":"23_CR17","doi-asserted-by":"publisher","first-page":"104924","DOI":"10.1016\/j.engappai.2022.104924","volume":"113","author":"T Jin","year":"2022","unstructured":"Jin, T., Yu, Q., Zhu, S., Lin, Z., Ren, J., Zhou, Y., Song, W.: Depth-aware gaze-following via auxiliary networks for robotics. Eng. Appl. Artif. Intell. 113, 104924 (2022)","journal-title":"Eng. Appl. Artif. Intell."},{"key":"23_CR18","doi-asserted-by":"crossref","unstructured":"Kellnhofer, P., Recasens, A., Stent, S., Matusik, W., Torralba, A.: Gaze360: physically unconstrained gaze estimation in the wild. In: International Conference on Computer Vision, pp. 6912\u20136921 (2019)","DOI":"10.1109\/ICCV.2019.00701"},{"key":"23_CR19","doi-asserted-by":"crossref","unstructured":"Kirillov, A., et\u00a0al.: Segment anything. In: International Conference on Computer Vision, pp. 4015\u20134026 (2023)","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"23_CR20","doi-asserted-by":"crossref","unstructured":"Li, F., et al.: Mask DINO: towards a unified transformer-based framework for object detection and segmentation. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 3041\u20133050 (2023)","DOI":"10.1109\/CVPR52729.2023.00297"},{"key":"23_CR21","doi-asserted-by":"crossref","unstructured":"Li, Y., Shen, W., Gao, Z., Zhu, Y., Zhai, G., Guo, G.: Looking here or there? Gaze following in 360-degree images. In: International Conference on Computer Vision, pp. 3742\u20133751 (2021)","DOI":"10.1109\/ICCV48922.2021.00372"},{"key":"23_CR22","doi-asserted-by":"crossref","unstructured":"Lian, D., Yu, Z., Gao, S.: Believe it or not, we know what you are looking at! In: Asian Conference on Computer Vision, pp. 35\u201350 (2018)","DOI":"10.1007\/978-3-030-20893-6_3"},{"key":"23_CR23","doi-asserted-by":"publisher","first-page":"4198","DOI":"10.1109\/TMM.2020.3038311","volume":"23","author":"K Lv","year":"2020","unstructured":"Lv, K., Sheng, H., Xiong, Z., Li, W., Zheng, L.: Improving driver gaze prediction with reinforced attention. IEEE Trans. Multimed. 23, 4198\u20134207 (2020)","journal-title":"IEEE Trans. Multimed."},{"issue":"1","key":"23_CR24","doi-asserted-by":"publisher","first-page":"654","DOI":"10.1038\/s41467-024-44824-z","volume":"15","author":"J Ma","year":"2024","unstructured":"Ma, J., He, Y., Li, F., Han, L., You, C., Wang, B.: Segment anything in medical images. Nat. Commun. 15(1), 654 (2024)","journal-title":"Nat. Commun."},{"key":"23_CR25","doi-asserted-by":"crossref","unstructured":"Miao, Q., Hoai, M., Samaras, D.: Patch-level gaze distribution prediction for gaze following. In: IEEE Winter Conference on Applications of Computer Vision, pp. 880\u2013889 (2023)","DOI":"10.1109\/WACV56688.2023.00094"},{"issue":"1","key":"23_CR26","doi-asserted-by":"publisher","first-page":"115","DOI":"10.1007\/BF02206861","volume":"20","author":"P Mundy","year":"1990","unstructured":"Mundy, P., Sigman, M., Kasari, C.: A longitudinal study of joint attention and language development in autistic children. J. Autism Dev. Disord. 20(1), 115\u2013128 (1990)","journal-title":"J. Autism Dev. Disord."},{"key":"23_CR27","doi-asserted-by":"crossref","unstructured":"Park, S., Spurr, A., Hilliges, O.: Deep pictorial gaze estimation. In: European Conference on Computer Vision, pp. 721\u2013738 (2018)","DOI":"10.1007\/978-3-030-01261-8_44"},{"key":"23_CR28","doi-asserted-by":"crossref","unstructured":"Park, S., Zhang, X., Bulling, A., Hilliges, O.: Learning to find eye region landmarks for remote gaze estimation in unconstrained settings. In: Proceedings of the ACM Symposium on Eye Tracking Research & Applications, pp. 1\u201310 (2018)","DOI":"10.1145\/3204493.3204545"},{"key":"23_CR29","unstructured":"Recasens, A., Khosla, A., Vondrick, C., Torralba, A.: Where are they looking? Adv. Neural Inform. Process. Syst. 28 (2015)"},{"key":"23_CR30","doi-asserted-by":"crossref","unstructured":"Recasens, A., Vondrick, C., Khosla, A., Torralba, A.: Following gaze in video. In: International Conference on Computer Vision, pp. 1435\u20131443 (2017)","DOI":"10.1109\/ICCV.2017.160"},{"key":"23_CR31","doi-asserted-by":"crossref","unstructured":"Ruzzi, A., et al.: GazeNeRF: 3D-aware gaze redirection with neural radiance fields. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 9676\u20139685 (2023)","DOI":"10.1109\/CVPR52729.2023.00933"},{"issue":"8","key":"23_CR32","doi-asserted-by":"publisher","first-page":"1204","DOI":"10.1016\/j.neubiorev.2009.06.001","volume":"33","author":"A Senju","year":"2009","unstructured":"Senju, A., Johnson, M.H.: Atypical eye contact in autism: models, mechanisms and development. Neurosci. Biobehav. Rev. 33(8), 1204\u20131214 (2009)","journal-title":"Neurosci. Biobehav. Rev."},{"key":"23_CR33","doi-asserted-by":"crossref","unstructured":"Shi, W., et al.: Real-time single image and video super-resolution using an efficient sub-pixel convolutional neural network. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 1874\u20131883 (2016)","DOI":"10.1109\/CVPR.2016.207"},{"key":"23_CR34","doi-asserted-by":"crossref","unstructured":"Tafasca, S., Gupta, A., Odobez, J.M.: ChildPlay: a new benchmark for understanding children\u2019s gaze behaviour. In: International Conference on Computer Vision, pp. 20935\u201320946 (2023)","DOI":"10.1109\/ICCV51070.2023.01914"},{"key":"23_CR35","doi-asserted-by":"crossref","unstructured":"Tomas, H., et al.: GOO: a dataset for gaze object prediction in retail environments. In: IEEE Conference on Computer Vision and Pattern Recognition Workshop, pp. 3125\u20133133 (2021)","DOI":"10.1109\/CVPRW53098.2021.00349"},{"key":"23_CR36","doi-asserted-by":"crossref","unstructured":"Tonini, F., Beyan, C., Ricci, E.: Multimodal across domains gaze target detection. In: International Conference on Multimodal Interact, pp. 420\u2013431 (2022)","DOI":"10.1145\/3536221.3556624"},{"key":"23_CR37","doi-asserted-by":"crossref","unstructured":"Tonini, F., Dall\u2019Asen, N., Beyan, C., Ricci, E.: Object-aware gaze target detection. In: International Conference on Computer Vision, pp. 21860\u201321869 (2023)","DOI":"10.1109\/ICCV51070.2023.01998"},{"key":"23_CR38","doi-asserted-by":"crossref","unstructured":"Tu, D., Min, X., Duan, H., Guo, G., Zhai, G., Shen, W.: End-to-end human-gaze-target detection with transformers. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 2192\u20132200 (2022)","DOI":"10.1109\/CVPR52688.2022.00224"},{"key":"23_CR39","doi-asserted-by":"crossref","unstructured":"Tu, D., Shen, W., Sun, W., Min, X., Zhai, G., Chen, C.: Un-Gaze: A unified transformer for joint gaze-location and gaze-object detection. IEEE Trans. Circuit Syst. Video Technol. (2023)","DOI":"10.1109\/TCSVT.2023.3318839"},{"key":"23_CR40","doi-asserted-by":"crossref","unstructured":"Wang, B., Guo, C., Jin, Y., Xia, H., Liu, N.: TransGOP: transformer-based gaze object prediction. In: AAAI Conference on Artificial Intelligence (2024)","DOI":"10.1609\/aaai.v38i9.28883"},{"key":"23_CR41","doi-asserted-by":"crossref","unstructured":"Wang, B., Hu, T., Li, B., Chen, X., Zhang, Z.: GaTector: a unified framework for gaze object prediction. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 19588\u201319597 (2022)","DOI":"10.1109\/CVPR52688.2022.01898"},{"key":"23_CR42","doi-asserted-by":"crossref","unstructured":"Wang, K., Ji, Q.: Real time eye gaze tracking with 3D deformable eye-face model. In: International Conference on Computer Vision, pp. 1003\u20131011 (2017)","DOI":"10.1109\/ICCV.2017.114"},{"issue":"1","key":"23_CR43","doi-asserted-by":"publisher","first-page":"219","DOI":"10.1109\/TCYB.2023.3244269","volume":"54","author":"X Wang","year":"2024","unstructured":"Wang, X., et al.: Dual regression-enhanced gaze target detection in the wild. IEEE Trans. Cybern. 54(1), 219\u2013229 (2024)","journal-title":"IEEE Trans. Cybern."},{"key":"23_CR44","doi-asserted-by":"crossref","unstructured":"Wang, Z., et\u00a0al.: Learning to detect head movement in unconstrained remote gaze estimation in the wild. In: IEEE Winter Conference on Applications of Computer Vision, pp. 3443\u20133452 (2020)","DOI":"10.1109\/WACV45572.2020.9093476"},{"key":"23_CR45","unstructured":"Zhang, H., et al.: DINO: DETR with improved denoising anchor boxes for end-to-end object detection. In: International Conference on Learning Representation (2022)"},{"key":"23_CR46","doi-asserted-by":"crossref","unstructured":"Zhang, X., Sugano, Y., Fritz, M., Bulling, A.: Appearance-based gaze estimation in the wild. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 4511\u20134520 (2015)","DOI":"10.1109\/CVPR.2015.7299081"},{"issue":"1","key":"23_CR47","doi-asserted-by":"publisher","first-page":"162","DOI":"10.1109\/TPAMI.2017.2778103","volume":"41","author":"X Zhang","year":"2017","unstructured":"Zhang, X., Sugano, Y., Fritz, M., Bulling, A.: MPIIGaze: Real-world dataset and deep appearance-based gaze estimation. IEEE Trans. Pattern Anal. Mach. Intell. 41(1), 162\u2013175 (2017)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"23_CR48","unstructured":"Zhao, X., et al.: Fast segment anything. arXiv preprint arXiv:2306.12156 (2023)"},{"key":"23_CR49","doi-asserted-by":"crossref","unstructured":"Zhu, W., Deng, H.: Monocular free-head 3D gaze tracking with deep learning and geometry constraints. In: International Conference on Computer Vision, pp. 3143\u20133152 (2017)","DOI":"10.1109\/ICCV.2017.341"},{"key":"23_CR50","unstructured":"Zhu, X., Su, W., Lu, L., Li, B., Wang, X., Dai, J.: Deformable DETR: deformable transformers for end-to-end object detection. In: International Conference on Learning Representation (2021)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72890-7_23","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,6]],"date-time":"2024-12-06T20:07:37Z","timestamp":1733515657000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72890-7_23"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,7]]},"ISBN":["9783031728891","9783031728907"],"references-count":50,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72890-7_23","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12,7]]},"assertion":[{"value":"7 December 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}