{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T17:42:45Z","timestamp":1777657365602,"version":"3.51.4"},"publisher-location":"Cham","reference-count":53,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031728969","type":"print"},{"value":"9783031728976","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T00:00:00Z","timestamp":1733097600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T00:00:00Z","timestamp":1733097600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72897-6_9","type":"book-chapter","created":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T21:35:58Z","timestamp":1733088958000},"page":"152-169","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Early Anticipation of\u00a0Driving Maneuvers"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4465-7066","authenticated-orcid":false,"given":"Abdul","family":"Wasi","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4448-5794","authenticated-orcid":false,"given":"Shankar","family":"Gangisetty","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7448-2271","authenticated-orcid":false,"given":"Shyam Nandan","family":"Rai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6767-7057","authenticated-orcid":false,"given":"C. V.","family":"Jawahar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,12,2]]},"reference":[{"key":"9_CR1","doi-asserted-by":"crossref","unstructured":"Aliakbarian, M.S., Saleh, F.S., Salzmann, M., Fernando, B., Petersson, L., Andersson, L.: Viena2: a driving anticipation dataset (2018)","DOI":"10.1007\/978-3-030-20887-5_28"},{"key":"9_CR2","doi-asserted-by":"crossref","unstructured":"Amadori, P.V., Fischer, T., Wang, R., Demiris, Y.: Decision anticipation for driving assistance systems. In: ITSC, pp.\u00a01\u20137. IEEE (2020)","DOI":"10.1109\/ITSC45102.2020.9294216"},{"key":"9_CR3","doi-asserted-by":"crossref","unstructured":"Arandjelovic, R., Zisserman, A.: Look, listen and learn. In: ICCV, pp. 609\u2013617 (2017)","DOI":"10.1109\/ICCV.2017.73"},{"key":"9_CR4","unstructured":"Carreira, J., Noland, E., Banki-Horvath, A., Hillier, C., Zisserman, A.: A short note about kinetics-600. CoRR (2018)"},{"key":"9_CR5","doi-asserted-by":"crossref","unstructured":"Damen, D., et\u00a0al.: Scaling egocentric vision: the epic-kitchens dataset. In: ECCV, pp. 720\u2013736 (2018)","DOI":"10.1007\/978-3-030-01225-0_44"},{"key":"9_CR6","doi-asserted-by":"crossref","unstructured":"Damen, D., et\u00a0al.: Rescaling egocentric vision: collection, pipeline and challenges for epic-kitchens-100. In: IJCV, pp. 1\u201323 (2022)","DOI":"10.1007\/s11263-021-01531-2"},{"key":"9_CR7","unstructured":"Dosovitskiy, A., et al.: An image is worth 16x16 words: transformers for image recognition at scale. In: ICLR (2021)"},{"key":"9_CR8","doi-asserted-by":"crossref","unstructured":"Dosovitskiy, A., et al.: Flownet: learning optical flow with convolutional networks. In: ICCV, pp. 2758\u20132766 (2015)","DOI":"10.1109\/ICCV.2015.316"},{"key":"9_CR9","doi-asserted-by":"crossref","unstructured":"Dutta, A., Zisserman, A.: The VIA annotation software for images, audio and video. In: ACM Multimedia, pp. 2276\u20132279 (2019)","DOI":"10.1145\/3343031.3350535"},{"key":"9_CR10","unstructured":"Fan, H., et al.: Multiscale vision transformers. In: ICCV, pp. 6824\u20136835 (2021)"},{"key":"9_CR11","doi-asserted-by":"crossref","unstructured":"Furnari, A., Farinella, G.M.: What would you expect? Anticipating egocentric actions with rolling-unrolling LSTMs and modality attention. In: ICCV, pp. 6252\u20136261 (2019)","DOI":"10.1109\/ICCV.2019.00635"},{"issue":"11","key":"9_CR12","doi-asserted-by":"publisher","first-page":"4021","DOI":"10.1109\/TPAMI.2020.2992889","volume":"43","author":"A Furnari","year":"2020","unstructured":"Furnari, A., Farinella, G.M.: Rolling-unrolling LSTMs for action anticipation from first-person video. IEEE TPAMI 43(11), 4021\u20134036 (2020)","journal-title":"IEEE TPAMI"},{"key":"9_CR13","doi-asserted-by":"crossref","unstructured":"Gao, J., Yang, Z., Nevatia, R.: RED: reinforced encoder-decoder networks for action anticipation. In: BMVC. BMVA Press (2017)","DOI":"10.5244\/C.31.92"},{"key":"9_CR14","doi-asserted-by":"crossref","unstructured":"Gebert, P., Roitberg, A., Haurilet, M., Stiefelhagen, R.: End-to-end prediction of driver intention using 3D convolutional neural networks. In: IEEE Intelligent Vehicles Symposium (IV), pp. 969\u2013974 (2019)","DOI":"10.1109\/IVS.2019.8814249"},{"key":"9_CR15","doi-asserted-by":"crossref","unstructured":"Girase, H., Agarwal, N., Choi, C., Mangalam, K.: Latency matters: real-time action forecasting transformer. In: CVPR, pp. 18759\u201318769 (2023)","DOI":"10.1109\/CVPR52729.2023.01799"},{"key":"9_CR16","doi-asserted-by":"crossref","unstructured":"Girdhar, R., Grauman, K.: Anticipative video transformer. In: ICCV, pp. 13505\u201313515 (2021)","DOI":"10.1109\/ICCV48922.2021.01325"},{"key":"9_CR17","doi-asserted-by":"crossref","unstructured":"Girdhar, R., Singh, M., Ravi, N., van\u00a0der Maaten, L., Joulin, A., Misra, I.: Omnivore: a single model for many visual modalities. In: CVPR, pp. 16102\u201316112 (2022)","DOI":"10.1109\/CVPR52688.2022.01563"},{"key":"9_CR18","doi-asserted-by":"crossref","unstructured":"Gong, D., Lee, J., Kim, M., Ha, S.J., Cho, M.: Future transformer for long-term action anticipation. In: CVPR, pp. 3052\u20133061 (2022)","DOI":"10.1109\/CVPR52688.2022.00306"},{"key":"9_CR19","doi-asserted-by":"crossref","unstructured":"Hara, K., Kataoka, H., Satoh, Y.: Can spatiotemporal 3D CNNs retrace the history of 2D CNNs and imagenet? In: CVPR, pp. 6546\u20136555 (2018)","DOI":"10.1109\/CVPR.2018.00685"},{"key":"9_CR20","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"489","DOI":"10.1007\/978-3-319-10584-0_32","volume-title":"Computer Vision \u2013 ECCV 2014","author":"D-A Huang","year":"2014","unstructured":"Huang, D.-A., Kitani, K.M.: Action-reaction: forecasting the dynamics of human interaction. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8695, pp. 489\u2013504. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10584-0_32"},{"key":"9_CR21","doi-asserted-by":"crossref","unstructured":"Jain, A., Koppula, H.S., Soh, S., Raghavan, B., Saxena, A.: Car that knows before you do: anticipating maneuvers via learning temporal driving models. In: ICCV (2015)","DOI":"10.1109\/ICCV.2015.364"},{"key":"9_CR22","doi-asserted-by":"crossref","unstructured":"Jain, A., Singh, A., Koppula, H.S., Soh, S., Saxena, A.: Recurrent neural networks for driver activity anticipation via sensory-fusion architecture. In: ICRA, pp. 3118\u20133125. IEEE (2016)","DOI":"10.1109\/ICRA.2016.7487478"},{"key":"9_CR23","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"126","DOI":"10.1007\/978-3-031-19778-9_8","volume-title":"ECCV 2022","author":"I Kasahara","year":"2022","unstructured":"Kasahara, I., Stent, S., Park, H.S.: Look both ways: Self-supervising driver gaze estimation and road scene saliency. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13673, pp. 126\u2013142. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19778-9_8"},{"issue":"4","key":"9_CR24","doi-asserted-by":"publisher","first-page":"714","DOI":"10.1109\/TIV.2020.3003889","volume":"5","author":"N Khairdoost","year":"2020","unstructured":"Khairdoost, N., Shirpour, M., Bauer, M.A., Beauchemin, S.S.: Real-time driver maneuver prediction using LSTM. IEEE Trans. Intell. Veh. 5(4), 714\u2013724 (2020)","journal-title":"IEEE Trans. Intell. Veh."},{"key":"9_CR25","doi-asserted-by":"crossref","unstructured":"Koppula, H.S., Saxena, A.: Anticipating human activities using object affordances for reactive robotic response. IEEE TPAMI 14\u201329 (2015)","DOI":"10.1109\/TPAMI.2015.2430335"},{"key":"9_CR26","doi-asserted-by":"crossref","unstructured":"Li, Y., et al.: Mvitv2: improved multiscale vision transformers for classification and detection. In: CVPR, pp. 4804\u20134814 (2022)","DOI":"10.1109\/CVPR52688.2022.00476"},{"key":"9_CR27","doi-asserted-by":"crossref","unstructured":"Liu, C., Chen, Y., Tai, L., Ye, H., Liu, M., Shi, B.E.: A gaze model improves autonomous driving. In: ACM Symposium on Eye Tracking Research & Applications, pp.\u00a01\u20135 (2019)","DOI":"10.1145\/3314111.3319846"},{"key":"9_CR28","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"704","DOI":"10.1007\/978-3-030-58452-8_41","volume-title":"Computer Vision \u2013 ECCV 2020","author":"M Liu","year":"2020","unstructured":"Liu, M., Tang, S., Li, Y., Rehg, J.M.: Forecasting human-object interaction: joint prediction of motor attention and\u00a0actions in first person video. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12346, pp. 704\u2013721. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58452-8_41"},{"key":"9_CR29","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. In: ICLR (Poster) (2019)"},{"key":"9_CR30","doi-asserted-by":"crossref","unstructured":"Ma, Y., et al.: Cemformer: learning to predict driver intentions from in-cabin and external cameras via spatial-temporal transformers. In: ITSC, pp. 4960\u20134966. IEEE (2023)","DOI":"10.1109\/ITSC57777.2023.10421798"},{"key":"9_CR31","doi-asserted-by":"crossref","unstructured":"Nagarajan, T., Li, Y., Feichtenhofer, C., Grauman, K.: Ego-topo: environment affordances from egocentric video. In: CVPR, pp. 163\u2013172 (2020)","DOI":"10.1109\/CVPR42600.2020.00024"},{"key":"9_CR32","doi-asserted-by":"crossref","unstructured":"Nawhal, M., Jyothi, A.A., Mori, G.: Rethinking learning approaches for long-term action anticipation. In: ECCV, pp. 558\u2013576 (2022)","DOI":"10.1007\/978-3-031-19830-4_32"},{"key":"9_CR33","doi-asserted-by":"crossref","unstructured":"Pal, A., Mondal, S., Christensen, H.I.: Looking at the right stuff-guided semantic-gaze for autonomous driving. In: CVPR, pp. 11883\u201311892 (2020)","DOI":"10.1109\/CVPR42600.2020.01190"},{"issue":"7","key":"9_CR34","doi-asserted-by":"publisher","first-page":"1720","DOI":"10.1109\/TPAMI.2018.2845370","volume":"41","author":"A Palazzi","year":"2018","unstructured":"Palazzi, A., Abati, D., Solera, F., Cucchiara, R., et al.: Predicting the driver\u2019s focus of attention: the DR (eye) VE project. IEEE TPAMI 41(7), 1720\u20131733 (2018)","journal-title":"IEEE TPAMI"},{"key":"9_CR35","doi-asserted-by":"crossref","unstructured":"Pang, B., Zha, K., Cao, H., Shi, C., Lu, C.: Deep RNN framework for visual sequential applications. In: CVPR, pp. 423\u2013432 (2019)","DOI":"10.1109\/CVPR.2019.00051"},{"key":"9_CR36","doi-asserted-by":"crossref","unstructured":"Ramanishka, V., Chen, Y.T., Misu, T., Saenko, K.: Toward driving scene understanding: a dataset for learning driver behavior and causal reasoning. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00803"},{"key":"9_CR37","doi-asserted-by":"crossref","unstructured":"Rong, Y., Akata, Z., Kasneci, E.: Driver intention anticipation based on in-cabin and driving scene monitoring. In: IEEE 23rd International Conference on Intelligent Transportation Systems (ITSC), pp.\u00a01\u20138 (2020)","DOI":"10.1109\/ITSC45102.2020.9294181"},{"key":"9_CR38","doi-asserted-by":"crossref","unstructured":"Sandler, M., Zhmoginov, A., Vladymyrov, M., Jackson, A.: Fine-tuning image transformers using learnable memory. In: CVPR, pp. 12155\u201312164 (2022)","DOI":"10.1109\/CVPR52688.2022.01184"},{"key":"9_CR39","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"154","DOI":"10.1007\/978-3-030-58517-4_10","volume-title":"Computer Vision \u2013 ECCV 2020","author":"F Sener","year":"2020","unstructured":"Sener, F., Singhania, D., Yao, A.: Temporal aggregate representations for long-range video understanding. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12361, pp. 154\u2013171. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58517-4_10"},{"key":"9_CR40","unstructured":"Shi, X., Chen, Z., Wang, H., Yeung, D.Y., Wong, W.K., Woo, W.C.: Convolutional LSTM network: a machine learning approach for precipitation nowcasting. In: NeurIPS, vol. 28 (2015)"},{"key":"9_CR41","unstructured":"Simonyan, K., Zisserman, A.: Two-stream convolutional networks for action recognition in videos. In: NeurIPS, vol. 27 (2014)"},{"key":"9_CR42","unstructured":"Somasundaram, K., et al.: Project aria: a new tool for egocentric multi-modal AI research. CoRR (2023)"},{"key":"9_CR43","doi-asserted-by":"crossref","unstructured":"Sun, C., Myers, A., Vondrick, C., Murphy, K., Schmid, C.: Videobert: a joint model for video and language representation learning. In: ICCV, pp. 7464\u20137473 (2019)","DOI":"10.1109\/ICCV.2019.00756"},{"key":"9_CR44","doi-asserted-by":"crossref","unstructured":"Tziafas, G., Kasaei, H.: Early or late fusion matters: efficient RGB-D fusion in vision transformers for 3D object recognition. In: IROS, pp. 9558\u20139565. IEEE (2023)","DOI":"10.1109\/IROS55552.2023.10341422"},{"key":"9_CR45","unstructured":"Vaswani, A., et al.: Attention is all you need. In: NeurIPS, vol. 30 (2017)"},{"key":"9_CR46","doi-asserted-by":"crossref","unstructured":"Vondrick, C., Pirsiavash, H., Torralba, A.: Anticipating visual representations from unlabeled video. In: CVPR, pp. 98\u2013106 (2016)","DOI":"10.1109\/CVPR.2016.18"},{"key":"9_CR47","doi-asserted-by":"crossref","unstructured":"Wu, C.Y., et al.: Memvit: memory-augmented multiscale vision transformer for efficient long-term video recognition. In: CVPR, pp. 13587\u201313597 (2022)","DOI":"10.1109\/CVPR52688.2022.01322"},{"key":"9_CR48","doi-asserted-by":"crossref","unstructured":"Wu, M., et al.: Gaze-based intention anticipation over driving manoeuvres in semi-autonomous vehicles. In: IROS, pp. 6210\u20136216. IEEE (2019)","DOI":"10.1109\/IROS40897.2019.8967779"},{"key":"9_CR49","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"658","DOI":"10.1007\/978-3-030-20873-8_42","volume-title":"Computer Vision \u2013 ACCV 2018","author":"Y Xia","year":"2019","unstructured":"Xia, Y., Zhang, D., Kim, J., Nakayama, K., Zipser, K., Whitney, D.: Predicting driver attention in critical situations. In: Jawahar, C.V., Li, H., Mori, G., Schindler, K. (eds.) ACCV 2018. LNCS, vol. 11365, pp. 658\u2013674. Springer, Cham (2019). https:\/\/doi.org\/10.1007\/978-3-030-20873-8_42"},{"key":"9_CR50","doi-asserted-by":"crossref","unstructured":"Yang, D., et\u00a0al.: Aide: a vision-driven multi-view, multi-modal, multi-tasking dataset for assistive driving perception. In: ICCV, pp. 20459\u201320470 (2023)","DOI":"10.1109\/ICCV51070.2023.01871"},{"key":"9_CR51","doi-asserted-by":"crossref","unstructured":"Zhong, Z., Schneider, D., Voit, M., Stiefelhagen, R., Beyerer, J.: Anticipative feature fusion transformer for multi-modal action anticipation. In: CVPR, pp. 6068\u20136077 (2023)","DOI":"10.1109\/WACV56688.2023.00601"},{"issue":"3","key":"9_CR52","first-page":"2284","volume":"23","author":"F Zhou","year":"2021","unstructured":"Zhou, F., Yang, X.J., De Winter, J.C.: Using eye-tracking data to predict situation awareness in real time during takeover transitions in conditionally automated driving. IEEE TITS 23(3), 2284\u20132295 (2021)","journal-title":"IEEE TITS"},{"key":"9_CR53","doi-asserted-by":"crossref","unstructured":"Zhu, X., Xiong, Y., Dai, J., Yuan, L., Wei, Y.: Deep feature flow for video recognition. In: CVPR, pp. 2349\u20132358 (2017)","DOI":"10.1109\/CVPR.2017.441"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72897-6_9","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T23:18:15Z","timestamp":1733095095000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72897-6_9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,2]]},"ISBN":["9783031728969","9783031728976"],"references-count":53,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72897-6_9","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12,2]]},"assertion":[{"value":"2 December 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}