{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T17:39:55Z","timestamp":1777657195873,"version":"3.51.4"},"publisher-location":"Cham","reference-count":48,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031726637","type":"print"},{"value":"9783031726644","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T00:00:00Z","timestamp":1729900800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T00:00:00Z","timestamp":1729900800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72664-4_12","type":"book-chapter","created":{"date-parts":[[2024,10,25]],"date-time":"2024-10-25T17:02:04Z","timestamp":1729875724000},"page":"205-222","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["HAT: History-Augmented Anchor Transformer for\u00a0Online Temporal Action Localization"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8491-0316","authenticated-orcid":false,"given":"Sakib","family":"Reza","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5012-5459","authenticated-orcid":false,"given":"Yuexi","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3201-6010","authenticated-orcid":false,"given":"Mohsen","family":"Moghaddam","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1945-9172","authenticated-orcid":false,"given":"Octavia","family":"Camps","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,10,26]]},"reference":[{"issue":"3","key":"12_CR1","doi-asserted-by":"publisher","first-page":"162","DOI":"10.1214\/ss\/1030037906","volume":"12","author":"J Aldrich","year":"1997","unstructured":"Aldrich, J.: Ra fisher and the making of maximum likelihood 1912\u20131922. Stat. Sci. 12(3), 162\u2013176 (1997)","journal-title":"Stat. Sci."},{"key":"12_CR2","doi-asserted-by":"crossref","unstructured":"An, J., Kang, H., Han, S.H., Yang, M.H., Kim, S.J.: Miniroad: minimal rnn framework for online action detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10341\u201310350 (2023)","DOI":"10.1109\/ICCV51070.2023.00949"},{"key":"12_CR3","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo vadis, action recognition? a new model and the kinetics dataset. In: proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6299\u20136308 (2017)","DOI":"10.1109\/CVPR.2017.502"},{"key":"12_CR4","doi-asserted-by":"crossref","unstructured":"Chen, J., Mittal, G., Yu, Y., Kong, Y., Chen, M.: Gatehub: gated history unit with background suppression for online action detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19925\u201319934 (2022)","DOI":"10.1109\/CVPR52688.2022.01930"},{"issue":"10","key":"12_CR5","doi-asserted-by":"publisher","first-page":"2723","DOI":"10.1109\/TMM.2019.2959977","volume":"22","author":"P Chen","year":"2019","unstructured":"Chen, P., Gan, C., Shen, G., Huang, W., Zeng, R., Tan, M.: Relation attention for temporal action localization. IEEE Trans. Multimedia 22(10), 2723\u20132733 (2019)","journal-title":"IEEE Trans. Multimedia"},{"key":"12_CR6","doi-asserted-by":"publisher","unstructured":"Cheng, F., Bertasius, G.: Tallformer: temporal action localization with a long-memory transformer. In: Avidan, S., Brostow, G., Cisse, M., Farinella, G.M., Hassner, T. (eds.) European Conference on Computer Vision, pp. 503\u2013521. Springer, Heidelberg (2022). https:\/\/doi.org\/10.1007\/978-3-031-19830-4_29","DOI":"10.1007\/978-3-031-19830-4_29"},{"key":"12_CR7","doi-asserted-by":"crossref","unstructured":"Damen, D., et\u00a0al.: Rescaling egocentric vision: collection, pipeline and challenges for epic-kitchens-100. Int. J. Comput. Vision 1\u201323 (2022)","DOI":"10.1007\/s11263-021-01531-2"},{"key":"12_CR8","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"269","DOI":"10.1007\/978-3-319-46454-1_17","volume-title":"Computer Vision \u2013 ECCV 2016","author":"R De Geest","year":"2016","unstructured":"De Geest, R., Gavves, E., Ghodrati, A., Li, Z., Snoek, C., Tuytelaars, T.: Online Action detection. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9909, pp. 269\u2013284. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46454-1_17"},{"key":"12_CR9","doi-asserted-by":"crossref","unstructured":"Eun, H., Moon, J., Park, J., Jung, C., Kim, C.: Learning to discriminate information for online action detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 809\u2013818 (2020)","DOI":"10.1109\/CVPR42600.2020.00089"},{"key":"12_CR10","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Malik, J., He, K.: Slowfast networks for video recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6202\u20136211 (2019)","DOI":"10.1109\/ICCV.2019.00630"},{"key":"12_CR11","doi-asserted-by":"crossref","unstructured":"Gao, J., Yang, Z., Chen, K., Sun, C., Nevatia, R.: Turn tap: temporal unit regression network for temporal action proposals. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 3628\u20133636 (2017)","DOI":"10.1109\/ICCV.2017.392"},{"key":"12_CR12","doi-asserted-by":"crossref","unstructured":"Gao, J., Yang, Z., Nevatia, R.: RED: reinforced encoder-decoder networks for action anticipation. In: British Machine Vision Conference 2017, BMVC 2017, London, UK, 4\u20137 September 2017. BMVA Press (2017)","DOI":"10.5244\/C.31.92"},{"key":"12_CR13","doi-asserted-by":"crossref","unstructured":"Gao, M., Xu, M., Davis, L.S., Socher, R., Xiong, C.: Startnet: online detection of action start in untrimmed videos. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5542\u20135551 (2019)","DOI":"10.1109\/ICCV.2019.00564"},{"key":"12_CR14","doi-asserted-by":"crossref","unstructured":"Guermal, M., Ali, A., Dai, R., Br\u00e9mond, F.: Joadaa: joint online action detection and action anticipation. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 6889\u20136898 (2024)","DOI":"10.1109\/WACV57701.2024.00674"},{"key":"12_CR15","doi-asserted-by":"publisher","unstructured":"Guo, H., Ren, Z., Wu, Y., Hua, G., Ji, Q.: Uncertainty-based spatial-temporal attention for online action detection. In: Avidan, S., Brostow, G., Cisse, M., Farinella, G.M., Hassner, T. (eds.) European Conference on Computer Vision, pp. 69\u201386. Springer, Heidelberg (2022). https:\/\/doi.org\/10.1007\/978-3-031-19772-7_5","DOI":"10.1007\/978-3-031-19772-7_5"},{"key":"12_CR16","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.cviu.2016.10.018","volume":"155","author":"H Idrees","year":"2017","unstructured":"Idrees, H., et al.: The thumos challenge on action recognition for videos \u201cin the wild\". Comput. Vis. Image Underst. 155, 1\u201323 (2017)","journal-title":"Comput. Vis. Image Underst."},{"key":"12_CR17","doi-asserted-by":"crossref","unstructured":"Kang, H., Kim, H., An, J., Cho, M., Kim, S.J.: Soft-landing strategy for alleviating the task discrepancy problem in temporal action localization tasks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6514\u20136523 (2023)","DOI":"10.1109\/CVPR52729.2023.00630"},{"key":"12_CR18","doi-asserted-by":"crossref","unstructured":"Kang, H., Kim, K., Ko, Y., Kim, S.J.: Cag-qil: context-aware actionness grouping via q imitation learning for online temporal action localization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13729\u201313738 (2021)","DOI":"10.1109\/ICCV48922.2021.01347"},{"key":"12_CR19","doi-asserted-by":"publisher","unstructured":"Kim, Y.H., Kang, H., Kim, S.J.: A sliding window scheme for online temporal action localization. In: Avidan, S., Brostow, G., Cisse, M., Farinella, G.M., Hassner, T. (eds.) European Conference on Computer Vision, pp. 653\u2013669. Springer, Heidelberg (2022). https:\/\/doi.org\/10.1007\/978-3-031-19830-4_37","DOI":"10.1007\/978-3-031-19830-4_37"},{"key":"12_CR20","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2021.107954","volume":"116","author":"YH Kim","year":"2021","unstructured":"Kim, Y.H., Nam, S., Kim, S.J.: Temporally smooth online action detection using cycle-consistent future anticipation. Pattern Recogn. 116, 107954 (2021)","journal-title":"Pattern Recogn."},{"key":"12_CR21","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2022.108871","volume":"131","author":"YH Kim","year":"2022","unstructured":"Kim, Y.H., Nam, S., Kim, S.J.: 2pesnet: towards online processing of temporal action localization. Pattern Recogn. 131, 108871 (2022)","journal-title":"Pattern Recogn."},{"key":"12_CR22","doi-asserted-by":"crossref","unstructured":"Li, B., et al.: Equalized focal loss for dense long-tailed object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6990\u20136999 (2022)","DOI":"10.1109\/CVPR52688.2022.00686"},{"key":"12_CR23","doi-asserted-by":"crossref","unstructured":"Li, Y., Liu, M., Rehg, J.M.: In the eye of beholder: joint learning of gaze and actions in first person video. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 619\u2013635 (2018)","DOI":"10.1007\/978-3-030-01228-1_38"},{"key":"12_CR24","doi-asserted-by":"crossref","unstructured":"Lin, C., et al.: Learning salient boundary feature for anchor-free temporal action localization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3320\u20133329 (2021)","DOI":"10.1109\/CVPR46437.2021.00333"},{"key":"12_CR25","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Goyal, P., Girshick, R., He, K., Doll\u00e1r, P.: Focal loss for dense object detection. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2980\u20132988 (2017)","DOI":"10.1109\/ICCV.2017.324"},{"key":"12_CR26","doi-asserted-by":"crossref","unstructured":"Liu, Q., Wang, Z.: Progressive boundary refinement network for temporal action detection. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a034, pp. 11612\u201311619 (2020)","DOI":"10.1609\/aaai.v34i07.6829"},{"key":"12_CR27","doi-asserted-by":"crossref","unstructured":"Liu, X., Hu, Y., Bai, S., Ding, F., Bai, X., Torr, P.H.: Multi-shot temporal event localization: a benchmark. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12596\u201312606 (2021)","DOI":"10.1109\/CVPR46437.2021.01241"},{"key":"12_CR28","doi-asserted-by":"crossref","unstructured":"Phan, T., Vo, K., Le, D., Doretto, G., Adjeroh, D., Le, N.: Zeetad: adapting pretrained vision-language model for zero-shot end-to-end temporal action detection. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 7046\u20137055 (2024)","DOI":"10.1109\/WACV57701.2024.00689"},{"key":"12_CR29","doi-asserted-by":"crossref","unstructured":"Shao, J., Wang, X., Quan, R., Zheng, J., Yang, J., Yang, Y.: Action sensitivity learning for temporal action localization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 13457\u201313469 (2023)","DOI":"10.1109\/ICCV51070.2023.01238"},{"key":"12_CR30","doi-asserted-by":"crossref","unstructured":"Shou, Z., et al.: Online detection of action start in untrimmed, streaming videos. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 534\u2013551 (2018)","DOI":"10.1007\/978-3-030-01219-9_33"},{"key":"12_CR31","doi-asserted-by":"crossref","unstructured":"Shou, Z., Wang, D., Chang, S.F.: Temporal action localization in untrimmed videos via multi-stage cnns. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1049\u20131058 (2016)","DOI":"10.1109\/CVPR.2016.119"},{"key":"12_CR32","doi-asserted-by":"crossref","unstructured":"Sukhbaatar, S., Grave, \u00c9., Bojanowski, P., Joulin, A.: Adaptive attention span in transformers. In: Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics, pp. 331\u2013335 (2019)","DOI":"10.18653\/v1\/P19-1032"},{"key":"12_CR33","unstructured":"Tang, T.N., Park, J., Kim, K., Sohn, K.: Simon: a simple framework for online temporal action localization. arXiv preprint arXiv:2211.04905 (2022)"},{"key":"12_CR34","doi-asserted-by":"crossref","unstructured":"Wang, J., Chen, G., Huang, Y., Wang, L., Lu, T.: Memory-and-anticipation transformer for online action understanding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13824\u201313835 (2023)","DOI":"10.1109\/ICCV51070.2023.01271"},{"key":"12_CR35","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"20","DOI":"10.1007\/978-3-319-46484-8_2","volume-title":"Computer Vision \u2013 ECCV 2016","author":"L Wang","year":"2016","unstructured":"Wang, L., Xiong, Y., Wang, Z., Qiao, Yu., Lin, D., Tang, X., Van Gool, L.: Temporal segment networks: towards good practices for deep action recognition. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9912, pp. 20\u201336. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46484-8_2"},{"key":"12_CR36","doi-asserted-by":"crossref","unstructured":"Wang, X., et al.: Oadtr: online action detection with transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7565\u20137575 (2021)","DOI":"10.1109\/ICCV48922.2021.00747"},{"key":"12_CR37","first-page":"9923","volume":"34","author":"M Xu","year":"2021","unstructured":"Xu, M., Perez Rua, J.M., Zhu, X., Ghanem, B., Martinez, B.: Low-fidelity video encoder optimization for temporal action localization. Adv. Neural. Inf. Process. Syst. 34, 9923\u20139935 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"12_CR38","doi-asserted-by":"crossref","unstructured":"Xu, M., Zhao, C., Rojas, D.S., Thabet, A., Ghanem, B.: G-tad: sub-graph localization for temporal action detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10156\u201310165 (2020)","DOI":"10.1109\/CVPR42600.2020.01017"},{"key":"12_CR39","doi-asserted-by":"crossref","unstructured":"Xu, M., Gao, M., Chen, Y.T., Davis, L.S., Crandall, D.J.: Temporal recurrent networks for online action detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5532\u20135541 (2019)","DOI":"10.1109\/ICCV.2019.00563"},{"key":"12_CR40","first-page":"1086","volume":"34","author":"M Xu","year":"2021","unstructured":"Xu, M., et al.: Long short-term transformer for online action detection. Adv. Neural. Inf. Process. Syst. 34, 1086\u20131099 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"12_CR41","doi-asserted-by":"crossref","unstructured":"Yang, L., Han, J., Zhang, D.: Colar: effective and efficient online action detection by consulting exemplars. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3160\u20133169 (2022)","DOI":"10.1109\/CVPR52688.2022.00316"},{"key":"12_CR42","doi-asserted-by":"crossref","unstructured":"Zeng, R., et al.: Graph convolutional networks for temporal action localization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7094\u20137103 (2019)","DOI":"10.1109\/ICCV.2019.00719"},{"key":"12_CR43","doi-asserted-by":"publisher","unstructured":"Zhang, C.L., Wu, J., Li, Y.: Actionformer: localizing moments of actions with transformers. In: Avidan, S., Brostow, G., Cisse, M., Farinella, G.M., Hassner, T. (eds.) European Conference on Computer Vision, pp. 492\u2013510. Springer, Heidelberg (2022). https:\/\/doi.org\/10.1007\/978-3-031-19772-7_29","DOI":"10.1007\/978-3-031-19772-7_29"},{"key":"12_CR44","doi-asserted-by":"crossref","unstructured":"Zhao, C., Liu, S., Mangalam, K., Ghanem, B.: Re2tal: rewiring pretrained video backbones for reversible temporal action localization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10637\u201310647 (2023)","DOI":"10.1109\/CVPR52729.2023.01025"},{"key":"12_CR45","doi-asserted-by":"crossref","unstructured":"Zhao, C., Thabet, A.K., Ghanem, B.: Video self-stitching graph network for temporal action localization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13658\u201313667 (2021)","DOI":"10.1109\/ICCV48922.2021.01340"},{"key":"12_CR46","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Xiong, Y., Wang, L., Wu, Z., Tang, X., Lin, D.: Temporal action detection with structured segment networks. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2914\u20132923 (2017)","DOI":"10.1109\/ICCV.2017.317"},{"key":"12_CR47","doi-asserted-by":"crossref","unstructured":"Zhao, Z., Wang, D., Zhao, X.: Movement enhancement toward multi-scale video feature representation for temporal action detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13555\u201313564 (2023)","DOI":"10.1109\/ICCV51070.2023.01247"},{"key":"12_CR48","doi-asserted-by":"crossref","unstructured":"Zhu, Z., Tang, W., Wang, L., Zheng, N., Hua, G.: Enriching local and global contexts for temporal action localization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13516\u201313525 (2021)","DOI":"10.1109\/ICCV48922.2021.01326"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72664-4_12","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,30]],"date-time":"2024-11-30T08:26:45Z","timestamp":1732955205000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72664-4_12"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,26]]},"ISBN":["9783031726637","9783031726644"],"references-count":48,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72664-4_12","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,26]]},"assertion":[{"value":"26 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}