{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T15:35:40Z","timestamp":1785512140602,"version":"3.56.0"},"publisher-location":"Cham","reference-count":51,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031197772","type":"print"},{"value":"9783031197789","type":"electronic"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-031-19778-9_8","type":"book-chapter","created":{"date-parts":[[2022,11,2]],"date-time":"2022-11-02T20:28:41Z","timestamp":1667420921000},"page":"126-142","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":38,"title":["Look Both\u00a0Ways: Self-supervising Driver Gaze Estimation and\u00a0Road Scene Saliency"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3964-1465","authenticated-orcid":false,"given":"Isaac","family":"Kasahara","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2623-6383","authenticated-orcid":false,"given":"Simon","family":"Stent","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6613-0738","authenticated-orcid":false,"given":"Hyun Soo","family":"Park","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,11,3]]},"reference":[{"key":"8_CR1","unstructured":"International Data Corporation: Worldwide Autonomous Vehicle Forecast, 2020\u20132024 (2020)"},{"key":"8_CR2","unstructured":"SAE Levels of Driving Automation Refined for Clarity and International Audience (2021). https:\/\/www.sae.org\/blog\/sae-j3016-update"},{"key":"8_CR3","doi-asserted-by":"crossref","unstructured":"Baee, S., Pakdamanian, E., Kim, I., Feng, L., Ordonez, V., Barnes, L.: MEDIRL: predicting the visual attention of drivers via maximum entropy deep inverse reinforcement learning. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.01293"},{"key":"8_CR4","unstructured":"Baluja, S., Pomerleau, D.: Non-intrusive gaze tracking using artificial neural networks (1993)"},{"key":"8_CR5","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"809","DOI":"10.1007\/978-3-319-46454-1_49","volume-title":"Computer Vision \u2013 ECCV 2016","author":"Z Bylinskii","year":"2016","unstructured":"Bylinskii, Z., Recasens, A., Borji, A., Oliva, A., Torralba, A., Durand, F.: Where should saliency models look next? In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9909, pp. 809\u2013824. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46454-1_49"},{"key":"8_CR6","doi-asserted-by":"crossref","unstructured":"Caesar, H., et al.: nuScenes: a multimodal dataset for autonomous driving. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.01164"},{"key":"8_CR7","doi-asserted-by":"crossref","unstructured":"Cao, Z., Hidalgo Martinez, G., Simon, T., Wei, S., Sheikh, Y.A.: OpenPose: realtime multi-person 2D pose estimation using part affinity fields. TPAMI 43, 172\u2013186 (2019)","DOI":"10.1109\/TPAMI.2019.2929257"},{"key":"8_CR8","doi-asserted-by":"crossref","unstructured":"Chang, Z., Matias Di Martino, J., Qiu, Q., Espinosa, S., Sapiro, G.: SalGaze: personalizing gaze estimation using visual saliency. In: ICCV Workshops (2019)","DOI":"10.1109\/ICCVW.2019.00148"},{"key":"8_CR9","doi-asserted-by":"crossref","unstructured":"Deng, H., Zhu, W.: Monocular free-head 3D gaze tracking with deep learning and geometry constraints. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.341"},{"key":"8_CR10","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"419","DOI":"10.1007\/978-3-030-58558-7_25","volume-title":"Computer Vision \u2013 ECCV 2020","author":"R Droste","year":"2020","unstructured":"Droste, R., Jiao, J., Noble, J.A.: Unified image and video saliency modeling. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12350, pp. 419\u2013435. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58558-7_25"},{"key":"8_CR11","doi-asserted-by":"crossref","unstructured":"Fang, J., Yan, D., Qiao, J., Xue, J., Yu, H.: DADA: driver attention prediction in driving accident scenarios. IEEE Trans. Intell. Transp. Syst. 23, 4959\u20134971 (2021)","DOI":"10.1109\/TITS.2020.3044678"},{"key":"8_CR12","doi-asserted-by":"crossref","unstructured":"Fischler, M.A., Bolles, R.C.: Random sample consensus: a paradigm for model fitting with applications to image analysis and automated cartography. ACM Commun. 24, 381\u2013395 (1981)","DOI":"10.1145\/358669.358692"},{"key":"8_CR13","doi-asserted-by":"crossref","unstructured":"Geiger, A., Lenz, P., Urtasun, R.: Are we ready for autonomous driving? The KITTI vision benchmark suite. In: CVPR (2012)","DOI":"10.1109\/CVPR.2012.6248074"},{"key":"8_CR14","doi-asserted-by":"crossref","unstructured":"Hansen, D.W., Ji, Q.: In the eye of the beholder: a survey of models for eyes and gaze. TPAMI 32, 478\u2013500 (2009)","DOI":"10.1109\/TPAMI.2009.30"},{"key":"8_CR15","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"8_CR16","doi-asserted-by":"crossref","unstructured":"Jain, A., Koppula, H.S., Raghavan, B., Soh, S., Saxena, A.: Car that knows before you do: anticipating maneuvers via learning temporal driving models. In: ICCV (2015)","DOI":"10.1109\/ICCV.2015.364"},{"key":"8_CR17","doi-asserted-by":"crossref","unstructured":"Jiang, M., Huang, S., Duan, J., Zhao, Q.: SALICON: saliency in context. In: CVPR (2015)","DOI":"10.1109\/CVPR.2015.7298710"},{"key":"8_CR18","doi-asserted-by":"crossref","unstructured":"Kellnhofer, P., Recasens, A., Stent, S., Matusik, W., Torralba, A.: Gaze360: physically unconstrained gaze estimation in the wild. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00701"},{"key":"8_CR19","unstructured":"Kingma, D.P., Ba, J.L.: Adam: a method for stochastic optimization. arXiv (2014)"},{"key":"8_CR20","doi-asserted-by":"crossref","unstructured":"Land, M.F.: Eye movements and the control of actions in everyday life. Prog. Retinal Eye Res. 25, 296\u2013324 (2006)","DOI":"10.1016\/j.preteyeres.2006.01.002"},{"key":"8_CR21","doi-asserted-by":"crossref","unstructured":"Lind\u00e9n, E., Sjostrand, J., Proutiere, A.: Learning to personalize in appearance-based gaze tracking. In: ICCV Workshops (2019)","DOI":"10.1109\/ICCVW.2019.00145"},{"key":"8_CR22","doi-asserted-by":"crossref","unstructured":"Lipson, L., Teed, Z., Deng, J.: RAFT-stereo: multilevel recurrent field transforms for stereo matching. In: 23DV (2021)","DOI":"10.1109\/3DV53792.2021.00032"},{"key":"8_CR23","doi-asserted-by":"crossref","unstructured":"Lowe, D.G.: Object recognition from local scale-invariant features. IJCV (1999)","DOI":"10.1109\/ICCV.1999.790410"},{"key":"8_CR24","doi-asserted-by":"crossref","unstructured":"Martin, M., et al.: Drive &Act: a multi-modal dataset for fine-grained driver behavior recognition in autonomous vehicles. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00289"},{"key":"8_CR25","doi-asserted-by":"crossref","unstructured":"Mathe, S., Sminchisescu, C.: Actions in the eye: dynamic gaze datasets and learnt saliency models for visual recognition. TPAMI 37, 1408\u20131424 (2015)","DOI":"10.1109\/TPAMI.2014.2366154"},{"key":"8_CR26","doi-asserted-by":"crossref","unstructured":"Min, K., Corso, J.J.: TASED-Net: temporally-aggregating spatial encoder-decoder network for video saliency detection. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00248"},{"key":"8_CR27","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"387","DOI":"10.1007\/978-3-030-66823-5_23","volume-title":"Computer Vision \u2013 ECCV 2020 Workshops","author":"JD Ortega","year":"2020","unstructured":"Ortega, J.D., et al.: DMD: a large-scale multi-modal driver monitoring dataset for attention and alertness analysis. In: Bartoli, A., Fusiello, A. (eds.) ECCV 2020. LNCS, vol. 12538, pp. 387\u2013405. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-66823-5_23"},{"key":"8_CR28","doi-asserted-by":"crossref","unstructured":"Palazzi, A., Abati, D., Solera, F., Cucchiara, R., et al.: Predicting the driver\u2019s focus of attention: the DR(eye)VE project. TPAMI 41, 1720\u20131733 (2018)","DOI":"10.1109\/TPAMI.2018.2845370"},{"key":"8_CR29","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"747","DOI":"10.1007\/978-3-030-58610-2_44","volume-title":"Computer Vision \u2013 ECCV 2020","author":"S Park","year":"2020","unstructured":"Park, S., Aksan, E., Zhang, X., Hilliges, O.: Towards end-to-end video-based eye-tracking. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12357, pp. 747\u2013763. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58610-2_44"},{"key":"8_CR30","doi-asserted-by":"crossref","unstructured":"Park, S., Mello, S.D., Molchanov, P., Iqbal, U., Hilliges, O., Kautz, J.: Few-shot adaptive gaze estimation. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00946"},{"key":"8_CR31","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"741","DOI":"10.1007\/978-3-030-01261-8_44","volume-title":"Computer Vision \u2013 ECCV 2018","author":"S Park","year":"2018","unstructured":"Park, S., Spurr, A., Hilliges, O.: Deep pictorial gaze estimation. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11217, pp. 741\u2013757. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01261-8_44"},{"key":"8_CR32","unstructured":"Paszke, A., et al.: PyTorch: an imperative style, high-performance deep learning library. In: NeurIPS (2019)"},{"key":"8_CR33","unstructured":"Recasens, A., Khosla, A., Vondrick, C., Torralba, A.: Where are they looking? In: NeurIPS (2015)"},{"key":"8_CR34","doi-asserted-by":"crossref","unstructured":"Recasens, A., Vondrick, C., Khosla, A., Torralba, A.: Following gaze in video. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.160"},{"key":"8_CR35","doi-asserted-by":"crossref","unstructured":"Sch\u00f6nberger, J.L., Frahm, J.M.: Structure-from-motion revisited. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.445"},{"key":"8_CR36","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"33","DOI":"10.1007\/978-3-319-10584-0_3","volume-title":"Computer Vision \u2013 ECCV 2014","author":"C Shen","year":"2014","unstructured":"Shen, C., Zhao, Q.: Webpage saliency. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8695, pp. 33\u201346. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10584-0_3"},{"key":"8_CR37","doi-asserted-by":"crossref","unstructured":"Shrivastava, A., Pfister, T., Tuzel, O., Susskind, J., Wang, W., Webb, R.: Learning from simulated and unsupervised images through adversarial training. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.241"},{"key":"8_CR38","doi-asserted-by":"crossref","unstructured":"Sugano, Y., Matsushita, Y., Sato, Y.: Appearance-based gaze estimation using visual saliency. TPAMI 35, 329\u2013341 (2013)","DOI":"10.1109\/TPAMI.2012.101"},{"key":"8_CR39","doi-asserted-by":"crossref","unstructured":"Sugano, Y., Matsushita, Y., Sato, Y.: Learning-by-synthesis for appearance-based 3D gaze estimation. In: CVPR (2014)","DOI":"10.1109\/CVPR.2014.235"},{"key":"8_CR40","doi-asserted-by":"crossref","unstructured":"Sun, P., et al.: Scalability in perception for autonomous driving: Waymo open dataset. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00252"},{"key":"8_CR41","doi-asserted-by":"crossref","unstructured":"Sun, Y., Zeng, J., Shan, S., Chen, X.: Cross-encoder for unsupervised gaze representation learning. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.00368"},{"key":"8_CR42","doi-asserted-by":"crossref","unstructured":"Wang, J., Olson, E.: AprilTag 2: efficient and robust fiducial detection. In: IROS (2016)","DOI":"10.1109\/IROS.2016.7759617"},{"key":"8_CR43","doi-asserted-by":"crossref","unstructured":"Wang, W., Shen, J., Xie, J., Cheng, M.M., Ling, H., Borji, A.: Revisiting video saliency prediction in the deep learning era. TPAMI 43, 220\u2013237 (2021)","DOI":"10.1109\/TPAMI.2019.2924417"},{"key":"8_CR44","doi-asserted-by":"crossref","unstructured":"Wood, E., Baltrusaitis, T., Zhang, X., Sugano, Y., Robinson, P., Bulling, A.: Rendering of eyes for eye-shape registration and gaze estimation. In: ICCV (2015)","DOI":"10.1109\/ICCV.2015.428"},{"key":"8_CR45","doi-asserted-by":"crossref","unstructured":"Wu, T., Martelaro, N., Stent, S., Ortiz, J., Ju, W.: Learning when agents can talk to drivers using the INAGT dataset and multisensor fusion. ACM Interact. Mob. Wearable Ubiquit. Technol. 5, 1\u201328 (2021)","DOI":"10.1145\/3478125"},{"key":"8_CR46","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"658","DOI":"10.1007\/978-3-030-20873-8_42","volume-title":"Computer Vision \u2013 ACCV 2018","author":"Y Xia","year":"2019","unstructured":"Xia, Y., Zhang, D., Kim, J., Nakayama, K., Zipser, K., Whitney, D.: Predicting driver attention in critical situations. In: Jawahar, C.V., Li, H., Mori, G., Schindler, K. (eds.) ACCV 2018. LNCS, vol. 11365, pp. 658\u2013674. Springer, Cham (2019). https:\/\/doi.org\/10.1007\/978-3-030-20873-8_42"},{"key":"8_CR47","doi-asserted-by":"publisher","unstructured":"Yarbus, A.L.: Eye Movements and Vision. Springer, New York (2013). https:\/\/doi.org\/10.1007\/978-1-4899-5379-7","DOI":"10.1007\/978-1-4899-5379-7"},{"key":"8_CR48","doi-asserted-by":"crossref","unstructured":"Yu, Y., Odobez, J.M.: Unsupervised representation learning for gaze estimation. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00734"},{"key":"8_CR49","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"365","DOI":"10.1007\/978-3-030-58558-7_22","volume-title":"Computer Vision \u2013 ECCV 2020","author":"X Zhang","year":"2020","unstructured":"Zhang, X., Park, S., Beeler, T., Bradley, D., Tang, S., Hilliges, O.: ETH-XGaze: a large scale dataset for gaze estimation under extreme head pose and gaze variation. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12350, pp. 365\u2013381. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58558-7_22"},{"key":"8_CR50","doi-asserted-by":"crossref","unstructured":"Zhang, X., Sugano, Y., Fritz, M., Bulling, A.: Appearance-based gaze estimation in the wild. In: CVPR (2015)","DOI":"10.1109\/CVPR.2015.7299081"},{"key":"8_CR51","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"300","DOI":"10.1007\/978-3-030-01264-9_18","volume-title":"Computer Vision \u2013 ECCV 2018","author":"Q Zheng","year":"2018","unstructured":"Zheng, Q., Jiao, J., Cao, Y., Lau, R.W.H.: Task-driven webpage saliency. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) Computer Vision \u2013 ECCV 2018. LNCS, vol. 11218, pp. 300\u2013316. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01264-9_18"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2022"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-19778-9_8","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,11,2]],"date-time":"2022-11-02T20:54:08Z","timestamp":1667422448000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-19778-9_8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783031197772","9783031197789"],"references-count":51,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-19778-9_8","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"3 November 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Tel Aviv","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Israel","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 October 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 October 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2022.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"CMT","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5804","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1645","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"28% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.21","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.91","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}