{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,9]],"date-time":"2026-01-09T20:37:28Z","timestamp":1767991048311,"version":"3.49.0"},"reference-count":52,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2023,9,26]],"date-time":"2023-09-26T00:00:00Z","timestamp":1695686400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,9,26]],"date-time":"2023-09-26T00:00:00Z","timestamp":1695686400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Machine Vision and Applications"],"published-print":{"date-parts":[[2023,11]]},"DOI":"10.1007\/s00138-023-01457-4","type":"journal-article","created":{"date-parts":[[2023,9,26]],"date-time":"2023-09-26T08:04:04Z","timestamp":1695715444000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["Cascading spatio-temporal attention network for real-time action detection"],"prefix":"10.1007","volume":"34","author":[{"given":"Jianhua","family":"Yang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ke","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ruifeng","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Petra","family":"Perner","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,9,26]]},"reference":[{"key":"1457_CR1","doi-asserted-by":"crossref","unstructured":"Arnab, A., Sun, C., Schmid, C.: Unified graph structured models for video understanding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8117\u20138126 (2021)","DOI":"10.1109\/ICCV48922.2021.00801"},{"key":"1457_CR2","unstructured":"Bertasius, G., Wang, H., Torresani, L.: Is space-time attention all you need for video understanding? In: ICML, p. 4 (2021)"},{"key":"1457_CR3","doi-asserted-by":"crossref","unstructured":"Cao, Y., Xu, J., Lin, S., Wei, F., Hu, H.: Gcnet: Non-local networks meet squeeze-excitation networks and beyond. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision Workshops (2019)","DOI":"10.1109\/ICCVW.2019.00246"},{"key":"1457_CR4","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo vadis, action recognition? A new model and the kinetics dataset. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6299\u20136308 (2017)","DOI":"10.1109\/CVPR.2017.502"},{"key":"1457_CR5","doi-asserted-by":"crossref","unstructured":"Chang, S., Wang, P., Wang, F., Li, H., Feng, J.: Augmented transformer with adaptive graph for temporal action proposal generation. arXiv:2103.16024 (2021)","DOI":"10.1145\/3552458.3556443"},{"key":"1457_CR6","doi-asserted-by":"crossref","unstructured":"Chao, Y.-W., Vijayanarasimhan, S., Seybold, B., Ross, D.A., Deng, J., Sukthankar, R.: Rethinking the faster r-cnn architecture for temporal action localization. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1130\u20131139 (2018)","DOI":"10.1109\/CVPR.2018.00124"},{"key":"1457_CR7","doi-asserted-by":"crossref","unstructured":"Chen, S., Sun, P., Xie, E., Ge, C., Wu, J., Ma, L., Shen, J., Luo, P.: Watch only once: An end-to-end video action detection framework. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8178\u20138187 (2021)","DOI":"10.1109\/ICCV48922.2021.00807"},{"issue":"5","key":"1457_CR8","doi-asserted-by":"publisher","first-page":"765","DOI":"10.1007\/s00138-018-0931-1","volume":"29","author":"Albert Clap\u00e9s","year":"2018","unstructured":"Clap\u00e9s, Albert, Pardo, \u00c0lex., Vila, Oriol Pujol, Escalera, Sergio: Action detection fusing multiple kinects and a wimu: an application to in-home assistive technology for the elderly. Mach. Vis. Appl. 29(5), 765\u2013788 (2018)","journal-title":"Mach. Vis. Appl."},{"key":"1457_CR9","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Pinz, A., Zisserman, A.: Convolutional two-stream network fusion for video action recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 1933\u20131941 (2016)","DOI":"10.1109\/CVPR.2016.213"},{"key":"1457_CR10","doi-asserted-by":"crossref","unstructured":"Gkioxari, G., Malik, J.: Finding action tubes. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 759\u2013768 (2015)","DOI":"10.1109\/CVPR.2015.7298676"},{"key":"1457_CR11","doi-asserted-by":"crossref","unstructured":"Gu, C., Sun, C., Ross, D.A., Vondrick, C., Pantofaru, C., Li, Y., Vijayanarasimhan, S., Toderici, G., Ricco, S., Sukthankar, R., et al.: Ava: a video dataset of spatio-temporally localized atomic visual actions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6047\u20136056 (2018)","DOI":"10.1109\/CVPR.2018.00633"},{"key":"1457_CR12","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"1457_CR13","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., Sun, G.: Squeeze-and-excitation networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7132\u20137141 (2018)","DOI":"10.1109\/CVPR.2018.00745"},{"key":"1457_CR14","doi-asserted-by":"crossref","unstructured":"Jhuang, H., Gall, J., Zuffi, S., Schmid, C., Black, M.J.: Towards understanding action recognition. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 3192\u20133199 (2013)","DOI":"10.1109\/ICCV.2013.396"},{"key":"1457_CR15","doi-asserted-by":"crossref","unstructured":"Kalogeiton, V., Weinzaepfel, P., Ferrari, V., Schmid, C.: Action tubelet detector for spatio-temporal action localization. In: Proceedings of the IEEE International Conference on Computer Vision, pp 4405\u20134413 (2017)","DOI":"10.1109\/ICCV.2017.472"},{"key":"1457_CR16","unstructured":"Kingma, D.P., Ba, J.: Adam: A method for stochastic optimization. arXiv:1412.6980 (2014)"},{"key":"1457_CR17","unstructured":"K\u00f6p\u00fckl\u00fc, O., Wei, X., Rigoll, G.: You only watch once: a unified cnn architecture for real-time spatiotemporal action localization. arXiv:1911.06644 (2019)"},{"key":"1457_CR18","doi-asserted-by":"crossref","unstructured":"Li, J., Liu, X., Zong, Z., Zhao, W., Zhang, M., Song, J.: Graph attention based proposal 3d convnets for action detection. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp. 4626\u20134633 (2020)","DOI":"10.1609\/aaai.v34i04.5893"},{"key":"1457_CR19","doi-asserted-by":"publisher","first-page":"2990","DOI":"10.1109\/TMM.2020.2965434","volume":"22","author":"J Li","year":"2020","unstructured":"Li, J., Liu, X., Zhang, W., Zhang, M., Song, J., Sebe, N.: Spatio-temporal attention networks for action recognition and detection. IEEE Trans. Multimed. 22, 2990\u20133001 (2020)","journal-title":"IEEE Trans. Multimed."},{"key":"1457_CR20","doi-asserted-by":"crossref","unstructured":"Li, Y., Wang, Z., Wang, L., Wu, G.: Actions as moving points. In: European Conference on Computer Vision, pp. 68\u201384. Springer (2020)","DOI":"10.1007\/978-3-030-58517-4_5"},{"key":"1457_CR21","doi-asserted-by":"crossref","unstructured":"Li, Y., Lin, W., See, J., Xu, N., Xu, S., Yan, K., Yang, C.: Cfad: Coarse-to-fine action detector for spatiotemporal action localization. In: European Conference on Computer Vision, pp. 510\u2013527. Springer (2020)","DOI":"10.1007\/978-3-030-58517-4_30"},{"key":"1457_CR22","doi-asserted-by":"publisher","first-page":"41","DOI":"10.1016\/j.cviu.2017.10.011","volume":"166","author":"Zhenyang Li","year":"2018","unstructured":"Li, Zhenyang, Gavrilyuk, Kirill, Gavves, Efstratios, Jain, Mihir, Snoek, Cees GM.: Videolstm convolves, attends and flows for action recognition. Comput. Vis. Image Underst. 166, 41\u201350 (2018)","journal-title":"Comput. Vis. Image Underst."},{"key":"1457_CR23","doi-asserted-by":"crossref","unstructured":"Lin, C., Xu, C., Luo, D., Wang, Y., Tai, Y., Wang, C., Li, J., Huang, F., Fu, Y.: Learning salient boundary feature for anchor-free temporal action localization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3320\u20133329 (2021)","DOI":"10.1109\/CVPR46437.2021.00333"},{"key":"1457_CR24","doi-asserted-by":"crossref","unstructured":"Lin, T., Zhao, X., Shou, Z.: Single shot temporal action detection. In: Proceedings of the 25th ACM international conference on Multimedia, pp. 988\u2013996 (2017)","DOI":"10.1145\/3123266.3123343"},{"key":"1457_CR25","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Goyal, P., Girshick, R., He, K., Doll\u00e1r, P.: Focal loss for dense object detection. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2980\u20132988 (2017)","DOI":"10.1109\/ICCV.2017.324"},{"key":"1457_CR26","doi-asserted-by":"crossref","unstructured":"Liu, W., Anguelov, D., Erhan, D., Szegedy, C., Reed, S., Fu, C.-Y., Berg, A.C.: Ssd: single shot multibox detector. In: European Conference on Computer Vision, pp. 21\u201337. Springer (2016)","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"1457_CR27","doi-asserted-by":"crossref","unstructured":"Ma, X., Luo, Z., Zhang, X., Liao, Q., Shen, X., Wang, M.: Spatio-temporal action detector with self-attention. In: 2021 International Joint Conference on Neural Networks (IJCNN), pp. 1\u20138. IEEE (2021)","DOI":"10.1109\/IJCNN52387.2021.9533300"},{"key":"1457_CR28","doi-asserted-by":"crossref","unstructured":"Pan, J., Chen, S., Shou, M., Zheng, L., Yu, S., Jing, L., Hongsheng: Actor-context-actor relation network for spatio-temporal action localization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 464\u2013474 (2021)","DOI":"10.1109\/CVPR46437.2021.00053"},{"key":"1457_CR29","doi-asserted-by":"crossref","unstructured":"Peng, X., Schmid, C.: Multi-region two-stream r-cnn for action detection. In: European Conference on Computer Vision, pp. 744\u2013759. Springer (2016)","DOI":"10.1007\/978-3-319-46493-0_45"},{"key":"1457_CR30","doi-asserted-by":"crossref","unstructured":"Ramaswamy, A., Seemakurthy, K., Gubbi, J., et al.: Video action re-localization using spatio-temporal correlation. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 192\u2013201 (2022)","DOI":"10.1109\/WACVW54805.2022.00025"},{"key":"1457_CR31","first-page":"91","volume":"28","author":"Shaoqing Ren","year":"2015","unstructured":"Ren, Shaoqing, He, Kaiming, Girshick, Ross, Sun, Jian: Faster r-cnn: towards real-time object detection with region proposal networks. Adv. Neural Inf. Process. Syst. 28, 91\u201399 (2015)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"1457_CR32","doi-asserted-by":"crossref","unstructured":"Saha, S., Singh, G., Sapienza, M., Torr, P.H.S., Cuzzolin, F.: Deep learning for detecting multiple space-time action tubes in videos. arXiv:1608.01529 (2016)","DOI":"10.5244\/C.30.58"},{"key":"1457_CR33","doi-asserted-by":"crossref","unstructured":"Singh, G., Saha, S., Sapienza, M., Torr, P.H.S., Cuzzolin, F.: Online real-time multiple spatiotemporal action localisation and prediction. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 3637\u20133646 (2017)","DOI":"10.1109\/ICCV.2017.393"},{"key":"1457_CR34","doi-asserted-by":"crossref","unstructured":"Song, L., Zhang, S., Yu, G., Sun, H.: Tacnet: transition-aware context network for spatio-temporal action detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11987\u201311995 (2019)","DOI":"10.1109\/CVPR.2019.01226"},{"key":"1457_CR35","unstructured":"Soomro, K., Zamir, A.R., Shah, M.: Ucf101: A dataset of 101 human actions classes from videos in the wild. arXiv:1212.0402 (2012)"},{"key":"1457_CR36","doi-asserted-by":"crossref","unstructured":"Sun, C., Shrivastava, A., Vondrick, C., Murphy, K., Sukthankar, R., Schmid, C.: Actor-centric relation network. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 318\u2013334 (2018)","DOI":"10.1007\/978-3-030-01252-6_20"},{"key":"1457_CR37","doi-asserted-by":"crossref","unstructured":"Ulutan, O., Rallapalli, S., Srivatsa, M., Torres, C., Manjunath, B.S.: Actor conditioned attention maps for video action detection. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 527\u2013536 (2020)","DOI":"10.1109\/WACV45572.2020.9093617"},{"key":"1457_CR38","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., Polosukhin, I.: Attention is all you need. In: Advances in Neural Information Processing Systems, pp. 5998\u20136008 (2017)"},{"issue":"6","key":"1457_CR39","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s00138-020-01089-y","volume":"31","author":"Michael Villamizar","year":"2020","unstructured":"Villamizar, Michael, Mart\u00ednez-Gonz\u00e1lez, A., Can\u00e9vet, Olivier, Odobez, J.-M.: Watchnet++: efficient and accurate depth-based network for detecting people attacks and intrusion. Mach. Vis. Appl. 31(6), 1\u201316 (2020)","journal-title":"Mach. Vis. Appl."},{"key":"1457_CR40","doi-asserted-by":"crossref","unstructured":"Wang, L., Qiao, Y., Tang, X., Van Gool, L.: Actionness estimation using hybrid fully convolutional networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2708\u20132717 (2016)","DOI":"10.1109\/CVPR.2016.296"},{"key":"1457_CR41","unstructured":"Wang, L., Yang, H., Wu, W., Yao, H., Huang, H.: Temporal action proposal generation with transformers. arXiv:2105.12043 (2021)"},{"key":"1457_CR42","doi-asserted-by":"crossref","unstructured":"Wang, X., Girshick, R., Gupta, A., He, K.: Non-local neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7794\u20137803 (2018)","DOI":"10.1109\/CVPR.2018.00813"},{"key":"1457_CR43","doi-asserted-by":"crossref","unstructured":"Weinzaepfel, P., Harchaoui, Z., Schmid, C.: Learning to track for spatio-temporal action localization. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 3164\u20133172 (2015)","DOI":"10.1109\/ICCV.2015.362"},{"key":"1457_CR44","doi-asserted-by":"crossref","unstructured":"Woo, S., Park, J., Lee, J.-Y., Kweon, I.S.: Cbam: Convolutional block attention module. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 3\u201319 (2018)","DOI":"10.1007\/978-3-030-01234-2_1"},{"key":"1457_CR45","doi-asserted-by":"crossref","unstructured":"Wu, J., Kuang, Z., Wang, L., Zhang, W., Wu, G.: Context-aware rcnn: a baseline for action detection in videos. In: European Conference on Computer Vision, pp. 440\u2013456. Springer (2020)","DOI":"10.1007\/978-3-030-58595-2_27"},{"key":"1457_CR46","doi-asserted-by":"crossref","unstructured":"Xu, H., Das, A., Saenko, K.: R-c3d: region convolutional 3d network for temporal activity detection. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 5783\u20135792 (2017)","DOI":"10.1109\/ICCV.2017.617"},{"issue":"1","key":"1457_CR47","doi-asserted-by":"publisher","first-page":"229","DOI":"10.1109\/TMM.2019.2924576","volume":"22","author":"Chenggang Yan","year":"2019","unstructured":"Yan, Chenggang, Yunbin, Tu., Wang, Xingzheng, Zhang, Yongbing, Hao, Xinhong, Zhang, Yongdong, Dai, Qionghai: Stat: apatial-temporal attention mechanism for video captioning. IEEE Trans. Multimed. 22(1), 229\u2013241 (2019)","journal-title":"IEEE Trans. Multimed."},{"key":"1457_CR48","doi-asserted-by":"publisher","first-page":"8535","DOI":"10.1109\/TIP.2020.3016486","volume":"29","author":"Le Yang","year":"2020","unstructured":"Yang, Le., Peng, Houwen, Zhang, Dingwen, Jianlong, Fu., Han, Junwei: Revisiting anchor mechanisms for temporal action localization. IEEE Trans. Image Process. 29, 8535\u20138548 (2020)","journal-title":"IEEE Trans. Image Process."},{"key":"1457_CR49","doi-asserted-by":"crossref","unstructured":"Yang, X., Yang, X., Liu, M.-Y., Xiao, F., Davis, L.S., Kautz, J.: Step: spatio-temporal progressive learning for video action detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 264\u2013272 (2019)","DOI":"10.1109\/CVPR.2019.00035"},{"key":"1457_CR50","doi-asserted-by":"crossref","unstructured":"Yu, F., Wang, D., Shelhamer, E., Darrell, T.: Deep layer aggregation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2403\u20132412 (2018)","DOI":"10.1109\/CVPR.2018.00255"},{"key":"1457_CR51","doi-asserted-by":"crossref","unstructured":"Zhao, J., Snoek, C.G.M.: Dance with flow: Two-in-one stream action detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9935\u20139944 (2019)","DOI":"10.1109\/CVPR.2019.01017"},{"key":"1457_CR52","unstructured":"Zhou, X., Wang, D., Kr\u00e4henb\u00fchl, P.: Objects as points. arXiv:1904.07850 (2019)"}],"container-title":["Machine Vision and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00138-023-01457-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00138-023-01457-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00138-023-01457-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,11,6]],"date-time":"2023-11-06T17:07:41Z","timestamp":1699290461000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00138-023-01457-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,9,26]]},"references-count":52,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2023,11]]}},"alternative-id":["1457"],"URL":"https:\/\/doi.org\/10.1007\/s00138-023-01457-4","relation":{},"ISSN":["0932-8092","1432-1769"],"issn-type":[{"value":"0932-8092","type":"print"},{"value":"1432-1769","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,9,26]]},"assertion":[{"value":"25 June 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 July 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 August 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 September 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"110"}}