{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T13:09:36Z","timestamp":1784552976670,"version":"3.55.0"},"reference-count":85,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2024,5,6]],"date-time":"2024-05-06T00:00:00Z","timestamp":1714953600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,5,6]],"date-time":"2024-05-06T00:00:00Z","timestamp":1714953600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2024,8]]},"DOI":"10.1007\/s11227-024-06138-1","type":"journal-article","created":{"date-parts":[[2024,5,6]],"date-time":"2024-05-06T16:01:56Z","timestamp":1715011316000},"page":"17952-17979","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["LGAFormer: transformer with local and global attention for action detection"],"prefix":"10.1007","volume":"80","author":[{"given":"Haiping","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fuxing","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dongjing","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xinhao","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dongjin","family":"Yu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Liming","family":"Guan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,5,6]]},"reference":[{"key":"6138_CR1","doi-asserted-by":"crossref","unstructured":"Zhao Y, Xiong Y, Wang L, Wu Z, Tang X, Lin D (2017) Temporal action detection with structured segment networks. In: Proceedings of the IEEE International Conference on Computer Vision, pp 2914\u20132923","DOI":"10.1109\/ICCV.2017.317"},{"key":"6138_CR2","doi-asserted-by":"crossref","unstructured":"Shou Z, Wang D, Chang S-F (2016) Temporal action localization in untrimmed videos via multi-stage cnns. 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp 1049\u20131058","DOI":"10.1109\/CVPR.2016.119"},{"key":"6138_CR3","doi-asserted-by":"crossref","unstructured":"Shou Z, Chan J, Zareian A, Miyazawa K, Chang S-F (2017) Cdc: convolutional-de-convolutional networks for precise temporal action localization in untrimmed videos. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp 1417\u20131426","DOI":"10.1109\/CVPR.2017.155"},{"key":"6138_CR4","doi-asserted-by":"crossref","unstructured":"Dai X, Singh B, Zhang G, Davis LS, Chen YQ (2017) Temporal context network for activity localization in videos. In: 2017 IEEE International Conference on Computer Vision (ICCV), pp 5727\u20135736","DOI":"10.1109\/ICCV.2017.610"},{"key":"6138_CR5","doi-asserted-by":"crossref","unstructured":"Liu Q, Wang Z (2020) Progressive boundary refinement network for temporal action detection. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol 34, pp 11612\u201311619","DOI":"10.1609\/aaai.v34i07.6829"},{"key":"6138_CR6","doi-asserted-by":"crossref","unstructured":"Xu H, Das A, Saenko K (2017) R-c3d: region convolutional 3d network for temporal activity detection. In: Proceedings of the IEEE International Conference on Computer Vision, pp 5783\u20135792","DOI":"10.1109\/ICCV.2017.617"},{"key":"6138_CR7","doi-asserted-by":"crossref","unstructured":"Tran D, Bourdev L, Fergus R, Torresani L, Paluri M (2015) Learning spatiotemporal features with 3d convolutional networks. In: Proceedings of the IEEE International Conference on Computer Vision, pp 4489\u20134497","DOI":"10.1109\/ICCV.2015.510"},{"key":"6138_CR8","unstructured":"Wang L, Yang H, Wu W, Yao H, Huang H (2021) Temporal action proposal generation with transformers. arXiv:2105.12043"},{"key":"6138_CR9","doi-asserted-by":"crossref","unstructured":"Cheng F, Bertasius G (2022) Tallformer: temporal action localization with long-memory transformer. In: European Conference on Computer Vision","DOI":"10.1007\/978-3-031-19830-4_29"},{"key":"6138_CR10","unstructured":"Li S, Zhang F, Zhao R-W, Feng R, Yang K, Liu L-N, Hou J (2022) Pyramid region-based slot attention network for temporal action proposal generation. In: British Machine Vision Conference"},{"key":"6138_CR11","doi-asserted-by":"crossref","unstructured":"Qing Z, Su H, Gan W, Wang D, Wu W, Wang X, Qiao Y, Yan J, Gao C, Sang N (2021) Temporal context aggregation network for temporal action proposal refinement. In: 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp 485\u2013494","DOI":"10.1109\/CVPR46437.2021.00055"},{"key":"6138_CR12","doi-asserted-by":"crossref","unstructured":"Weng Y, Pan Z, Han M, Chang X, Zhuang B (2022) An efficient spatio-temporal pyramid transformer for action detection. In: Proceedings of Computer Vision\u2013ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23\u201327, 2022, Part XXXIV. Springer, pp. 358\u2013375","DOI":"10.1007\/978-3-031-19830-4_21"},{"key":"6138_CR13","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. In: Advances in Neural Information Processing Systems, vol 30"},{"key":"6138_CR14","doi-asserted-by":"crossref","unstructured":"Li Y, Mao H, Girshick R, He K (2022) Exploring plain vision transformer backbones for object detection. In: European Conference on Computer Vision. Springer, pp 280\u2013296","DOI":"10.1007\/978-3-031-20077-9_17"},{"key":"6138_CR15","doi-asserted-by":"crossref","unstructured":"Li Y, Wu C-Y, Fan H, Mangalam K, Xiong B, Malik J, Feichtenhofer C (2022) Mvitv2: improved multiscale vision transformers for classification and detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 4804\u20134814","DOI":"10.1109\/CVPR52688.2022.00476"},{"key":"6138_CR16","doi-asserted-by":"crossref","unstructured":"Carion N, Massa F, Synnaeve G, Usunier N, Kirillov A, Zagoruyko S (2020) End-to-end object detection with transformers. In: European Conference on Computer Vision. Springer, pp 213\u2013229","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"6138_CR17","doi-asserted-by":"crossref","unstructured":"Liu Z, Lin Y, Cao Y, Hu H, Wei Y, Zhang Z, Lin S, Guo B (2021) Swin transformer: hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 10012\u201310022","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"6138_CR18","doi-asserted-by":"crossref","unstructured":"Ding M, Xiao B, Codella N, Luo P, Wang J, Yuan L (2022) Davit: dual attention vision transformers. In: European Conference on Computer Vision. Springer, pp 74\u201392","DOI":"10.1007\/978-3-031-20053-3_5"},{"key":"6138_CR19","unstructured":"Tong Z, Song Y, Wang J, Wang L (2022) Videomae: masked autoencoders are data-efficient learners for self-supervised video pre-training. In: Advances in Neural Information Processing Systems, vol 35, pp 10078\u201310093"},{"key":"6138_CR20","unstructured":"Li K, Wang Y, He Y, Li Y, Wang Y, Wang L, Qiao Y (2022) Uniformerv2: spatiotemporal learning by arming image vits with video uniformer. arXiv:2211.09552"},{"key":"6138_CR21","doi-asserted-by":"crossref","unstructured":"Yan S, Xiong X, Arnab A, Lu Z, Zhang M, Sun C, Schmid C (2022) Multiview transformers for video recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 3333\u20133343","DOI":"10.1109\/CVPR52688.2022.00333"},{"key":"6138_CR22","doi-asserted-by":"crossref","unstructured":"Liu Z, Ning J, Cao Y, Wei Y, Zhang Z, Lin S, Hu H (2022) Video swin transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 3202\u20133211","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"6138_CR23","doi-asserted-by":"crossref","unstructured":"Qing Z, Zhang S, Huang Z, Wang X, Wang Y, Lv Y, Gao C, Sang N (2023) Mar: masked autoencoders for efficient action recognition. IEEE Trans Multimed","DOI":"10.1109\/TMM.2023.3263288"},{"key":"6138_CR24","doi-asserted-by":"crossref","unstructured":"Dai R, Das S, Kahatapitiya K, Ryoo MS, Br\u00e9mond F (2022) Ms-tct: multi-scale temporal convtransformer for action detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 20041\u201320051","DOI":"10.1109\/CVPR52688.2022.01941"},{"key":"6138_CR25","unstructured":"Tolstikhin IO, Houlsby N, Kolesnikov A, Beyer L, Zhai X, Unterthiner T, Yung J, Steiner A, Keysers D, Uszkoreit J (2021) Mlp-mixer: an all-mlp architecture for vision. In: Advances in Neural Information Processing Systems, vol 34, pp 24261\u201324272"},{"key":"6138_CR26","doi-asserted-by":"crossref","unstructured":"Yu W, Luo M, Zhou P, Si C, Zhou Y, Wang X, Feng J, Yan S (2022) Metaformer is actually what you need for vision. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 10819\u201310829","DOI":"10.1109\/CVPR52688.2022.01055"},{"key":"6138_CR27","doi-asserted-by":"crossref","unstructured":"Shi D, Zhong Y, Cao Q, Ma L, Li J, Tao D (2023) Tridet: temporal action detection with relative boundary modeling. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 18857\u201318866","DOI":"10.1109\/CVPR52729.2023.01808"},{"key":"6138_CR28","doi-asserted-by":"publisher","DOI":"10.1016\/j.media.2022.102676","volume":"83","author":"S Basu","year":"2023","unstructured":"Basu S, Gupta M, Rana P, Gupta P, Arora C (2023) Radformer: transformers with global-local attention for interpretable and accurate gallbladder cancer detection. Med Image Anal 83:102676","journal-title":"Med Image Anal"},{"issue":"1","key":"6138_CR29","doi-asserted-by":"publisher","first-page":"305","DOI":"10.1007\/s40747-023-01171-8","volume":"10","author":"GA Kumie","year":"2024","unstructured":"Kumie GA, Habtie MA, Ayall TA, Zhou C, Liu H, Seid AM, Erbad A (2024) Dual-attention network for view-invariant action recognition. Complex Intell Syst 10(1):305\u2013321","journal-title":"Complex Intell Syst"},{"key":"6138_CR30","doi-asserted-by":"publisher","first-page":"5427","DOI":"10.1109\/TIP.2022.3195321","volume":"31","author":"X Liu","year":"2022","unstructured":"Liu X, Wang Q, Hu Y, Tang X, Zhang S, Bai S, Bai X (2022) End-to-end temporal action detection with transformer. IEEE Trans Image Process 31:5427\u20135441","journal-title":"IEEE Trans Image Process"},{"key":"6138_CR31","unstructured":"Raghu M, Unterthiner T, Kornblith S, Zhang C, Dosovitskiy A (2021) Do vision transformers see like convolutional neural networks? In: Advances in neural information processing systems, vol 34, pp 12116\u201312128"},{"key":"6138_CR32","doi-asserted-by":"crossref","unstructured":"Wu Z, Su L, Huang Q (2019) Cascaded partial decoder for fast and accurate salient object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 3907\u20133916","DOI":"10.1109\/CVPR.2019.00403"},{"key":"6138_CR33","doi-asserted-by":"crossref","unstructured":"Hou Q, Cheng M-M, Hu X, Borji A, Tu Z, Torr PH (2017) Deeply supervised salient object detection with short connections. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 3203\u20133212","DOI":"10.1109\/CVPR.2017.563"},{"key":"6138_CR34","doi-asserted-by":"crossref","unstructured":"Zhang C-L, Wu J, Li Y (2022) Actionformer: localizing moments of actions with transformers. In: European Conference on Computer Vision. Springer, pp 492\u2013510","DOI":"10.1007\/978-3-031-19772-7_29"},{"key":"6138_CR35","doi-asserted-by":"crossref","unstructured":"Lin C, Li J, Wang Y, Tai Y, Luo D, Cui Z, Wang C, Li J, Huang F, Ji R (2020) Fast learning of temporal action proposal via dense boundary generator. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol 34, pp 11499\u201311506","DOI":"10.1609\/aaai.v34i07.6815"},{"key":"6138_CR36","doi-asserted-by":"publisher","first-page":"8535","DOI":"10.1109\/TIP.2020.3016486","volume":"29","author":"L Yang","year":"2020","unstructured":"Yang L, Peng H, Zhang D, Fu J, Han J (2020) Revisiting anchor mechanisms for temporal action localization. IEEE Trans Image Process 29:8535\u20138548","journal-title":"IEEE Trans Image Process"},{"key":"6138_CR37","doi-asserted-by":"crossref","unstructured":"Lin C, Xu C, Luo D, Wang Y, Tai Y, Wang C, Li J, Huang F, Fu Y (2021) Learning salient boundary feature for anchor-free temporal action localization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 3320\u20133329","DOI":"10.1109\/CVPR46437.2021.00333"},{"key":"6138_CR38","doi-asserted-by":"crossref","unstructured":"Chen G, Zheng Y-D, Wang L, Lu T (2022) Dcan: improving temporal action detection via dual context aggregation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol 36, pp 248\u2013257","DOI":"10.1609\/aaai.v36i1.19900"},{"key":"6138_CR39","doi-asserted-by":"crossref","unstructured":"Liu X, Bai S, Bai X (2022) An empirical study of end-to-end temporal action detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 20010\u201320019","DOI":"10.1109\/CVPR52688.2022.01938"},{"issue":"11","key":"6138_CR40","doi-asserted-by":"publisher","first-page":"2740","DOI":"10.1109\/TPAMI.2018.2868668","volume":"41","author":"L Wang","year":"2018","unstructured":"Wang L, Xiong Y, Wang Z, Qiao Y, Lin D, Tang X, Van Gool L (2018) Temporal segment networks for action recognition in videos. IEEE Trans Pattern Anal Mach Intell 41(11):2740\u20132755","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"6138_CR41","doi-asserted-by":"crossref","unstructured":"Carreira J, Zisserman A (2017) Quo vadis, action recognition? A new model and the kinetics dataset. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 6299\u20136308","DOI":"10.1109\/CVPR.2017.502"},{"key":"6138_CR42","doi-asserted-by":"crossref","unstructured":"Feichtenhofer C, Fan H, Malik J, He K (2019) Slowfast networks for video recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 6202\u20136211","DOI":"10.1109\/ICCV.2019.00630"},{"key":"6138_CR43","doi-asserted-by":"crossref","unstructured":"Shou Z, Wang D, Chang S-F (2016) Temporal action localization in untrimmed videos via multi-stage cnns. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 1049\u20131058","DOI":"10.1109\/CVPR.2016.119"},{"key":"6138_CR44","doi-asserted-by":"crossref","unstructured":"Tan J, Tang J, Wang L, Wu G (2021) Relaxed transformer decoders for direct action proposal generation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 13526\u201313535","DOI":"10.1109\/ICCV48922.2021.01327"},{"key":"6138_CR45","doi-asserted-by":"crossref","unstructured":"Bai Y, Wang Y, Tong Y, Yang Y, Liu Q, Liu J (2020) Boundary content graph neural network for temporal action proposal generation. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XXVIII vol 16. Springer, pp 121\u2013137","DOI":"10.1007\/978-3-030-58604-1_8"},{"key":"6138_CR46","doi-asserted-by":"crossref","unstructured":"Xu M, Zhao C, Rojas DS, Thabet A, Ghanem B (2020) G-tad: sub-graph localization for temporal action detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 10156\u201310165","DOI":"10.1109\/CVPR42600.2020.01017"},{"key":"6138_CR47","doi-asserted-by":"crossref","unstructured":"Su H, Gan W, Wu W, Qiao Y, Yan J (2021) Bsn++: complementary boundary regressor with scale-balanced relation modeling for temporal action proposal generation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol 35, pp 2602\u20132610","DOI":"10.1609\/aaai.v35i3.16363"},{"key":"6138_CR48","doi-asserted-by":"crossref","unstructured":"Sridhar D, Quader N, Muralidharan S, Li Y, Dai P, Lu J (2021) Class semantics-based attention for action detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 13739\u201313748","DOI":"10.1109\/ICCV48922.2021.01348"},{"key":"6138_CR49","doi-asserted-by":"crossref","unstructured":"Zhao C, Thabet AK, Ghanem B (2021) Video self-stitching graph network for temporal action localization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 13658\u201313667","DOI":"10.1109\/ICCV48922.2021.01340"},{"issue":"8","key":"6138_CR50","doi-asserted-by":"publisher","first-page":"8322","DOI":"10.1007\/s11227-022-04973-8","volume":"79","author":"X Liao","year":"2023","unstructured":"Liao X, Yuan J, Cai Z, Lai J-h (2023) An attention-based bidirectional gru network for temporal action proposals generation. J Supercomput 79(8):8322\u20138339","journal-title":"J Supercomput"},{"key":"6138_CR51","doi-asserted-by":"crossref","unstructured":"Lin T, Zhao X, Shou Z (2017) Single shot temporal action detection. In: Proceedings of the 25th ACM International Conference on Multimedia, pp 988\u2013996","DOI":"10.1145\/3123266.3123343"},{"key":"6138_CR52","doi-asserted-by":"crossref","unstructured":"Tian Z, Shen C, Chen H, He T (2019) Fcos: fully convolutional one-stage object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 9627\u20139636","DOI":"10.1109\/ICCV.2019.00972"},{"key":"6138_CR53","doi-asserted-by":"crossref","unstructured":"Liu W, Anguelov D, Erhan D, Szegedy C, Reed S, Fu C-Y, Berg AC (2016) Ssd: single shot multibox detector. In: Proceedings of Computer Vision\u2013ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11\u201314, 2016, Part I vol 14. Springer, pp 21\u201337","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"6138_CR54","doi-asserted-by":"crossref","unstructured":"Law H, Deng J (2018) Cornernet: detecting objects as paired keypoints. In: Proceedings of the European Conference on Computer Vision (ECCV), pp 734\u2013750","DOI":"10.1007\/978-3-030-01264-9_45"},{"key":"6138_CR55","doi-asserted-by":"publisher","first-page":"7389","DOI":"10.1109\/TIP.2020.3002345","volume":"29","author":"T Kong","year":"2020","unstructured":"Kong T, Sun F, Liu H, Jiang Y, Li L, Shi J (2020) Foveabox: beyound anchor-based object detection. IEEE Trans Image Process 29:7389\u20137398","journal-title":"IEEE Trans Image Process"},{"key":"6138_CR56","doi-asserted-by":"crossref","unstructured":"Zhu C, He Y, Savvides M (2019) Feature selective anchor-free module for single-shot object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 840\u2013849","DOI":"10.1109\/CVPR.2019.00093"},{"key":"6138_CR57","doi-asserted-by":"crossref","unstructured":"Redmon J, Divvala S, Girshick R, Farhadi A (2016) You only look once: unified, real-time object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 779\u2013788","DOI":"10.1109\/CVPR.2016.91"},{"key":"6138_CR58","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A, Weissenborn D, Zhai X, Unterthiner T, Dehghani M, Minderer M, Heigold G, Gelly S, et al (2020) An image is worth 16x16 words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929"},{"key":"6138_CR59","doi-asserted-by":"crossref","unstructured":"Yang J, Dong X, Liu L, Zhang C, Shen J, Yu D (2022) Recurring the transformer for video action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 14063\u201314073","DOI":"10.1109\/CVPR52688.2022.01367"},{"key":"6138_CR60","unstructured":"Bulat A, Perez Rua JM, Sudhakaran S, Martinez B, Tzimiropoulos G (2021) Space-time mixing attention for video transformer. In: Advances in Neural Information Processing Systems, vol 34, pp 19594\u201319607"},{"key":"6138_CR61","doi-asserted-by":"crossref","unstructured":"Arnab A, Dehghani M, Heigold G, Sun C, Lu\u010di\u0107 M, Schmid C (2021) Vivit: a video vision transformer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 6836\u20136846","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"6138_CR62","unstructured":"Bertasius G, Wang H, Torresani L (2021) Is space-time attention all you need for video understanding? In: ICML, vol 2, p 4"},{"key":"6138_CR63","doi-asserted-by":"crossref","unstructured":"Zhao P, Xie L, Ju C, Zhang Y, Wang Y, Tian Q (2020) Bottom-up temporal action localization with mutual regularization. In: Proceedings of Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Part VIII 16. Springer, pp 539\u2013555","DOI":"10.1007\/978-3-030-58598-3_32"},{"key":"6138_CR64","doi-asserted-by":"crossref","unstructured":"Liu D, Jiang T, Wang Y (2019) Completeness modeling and context separation for weakly supervised temporal action localization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 1298\u20131307","DOI":"10.1109\/CVPR.2019.00139"},{"key":"6138_CR65","unstructured":"Choromanski K, Likhosherstov V, Dohan D, Song X, Gane A, Sarlos T, Hawkins P, Davis J, Mohiuddin A, Kaiser L, et al (2020) Rethinking attention with performers. arXiv:2009.14794"},{"key":"6138_CR66","doi-asserted-by":"crossref","unstructured":"Lin T-Y, Goyal P, Girshick R, He K, Doll\u00e1r P (2017) Focal loss for dense object detection. In: Proceedings of the IEEE International Conference on Computer Vision, pp 2980\u20132988","DOI":"10.1109\/ICCV.2017.324"},{"key":"6138_CR67","doi-asserted-by":"crossref","unstructured":"Rezatofighi H, Tsoi N, Gwak J, Sadeghian A, Reid I, Savarese S (2019) Generalized intersection over union: a metric and a loss for bounding box regression. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 658\u2013666","DOI":"10.1109\/CVPR.2019.00075"},{"key":"6138_CR68","doi-asserted-by":"crossref","unstructured":"Tian Z, Shen C, Chen H, He T (2019) Fcos: fully convolutional one-stage object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 9627\u20139636","DOI":"10.1109\/ICCV.2019.00972"},{"key":"6138_CR69","doi-asserted-by":"crossref","unstructured":"Zhang S, Chi C, Yao Y, Lei Z, Li SZ (2020) Bridging the gap between anchor-based and anchor-free detection via adaptive training sample selection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 9759\u20139768","DOI":"10.1109\/CVPR42600.2020.00978"},{"key":"6138_CR70","doi-asserted-by":"crossref","unstructured":"Bodla N, Singh B, Chellappa R, Davis LS (2017) Soft-nms-improving object detection with one line of code. In: Proceedings of the IEEE International Conference on Computer Vision, pp 5561\u20135569","DOI":"10.1109\/ICCV.2017.593"},{"key":"6138_CR71","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.cviu.2016.10.018","volume":"155","author":"H Idrees","year":"2017","unstructured":"Idrees H, Zamir AR, Jiang Y-G, Gorban A, Laptev I, Sukthankar R, Shah M (2017) The thumos challenge on action recognition for videos in the wild. Comput Vis Image Underst 155:1\u201323","journal-title":"Comput Vis Image Underst"},{"key":"6138_CR72","doi-asserted-by":"crossref","unstructured":"Caba Heilbron F, Escorcia V, Ghanem B, Carlos Niebles J (2015) Activitynet: a large-scale video benchmark for human activity understanding. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 961\u2013970","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"6138_CR73","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s11263-021-01531-2","volume":"130","author":"D Damen","year":"2022","unstructured":"Damen D, Doughty H, Farinella GM, Furnari A, Kazakos E, Ma J, Moltisanti D, Munro J, Perrett T, Price W et al (2022) Rescaling egocentric vision: collection, pipeline and challenges for epic-kitchens-100. Int J Comput Vis 130:1\u201323","journal-title":"Int J Comput Vis"},{"key":"6138_CR74","unstructured":"Loshchilov I, Hutter F (2017) Decoupled weight decay regularization. arXiv:1711.05101"},{"key":"6138_CR75","doi-asserted-by":"crossref","unstructured":"Lin T, Zhao X, Su H, Wang C, Yang M (2018) Bsn: boundary sensitive network for temporal action proposal generation. In: Proceedings of the European Conference on Computer Vision (ECCV), pp 3\u201319","DOI":"10.1007\/978-3-030-01225-0_1"},{"key":"6138_CR76","doi-asserted-by":"crossref","unstructured":"Lin T, Liu X, Li X, Ding E, Wen S (2019) Bmn: boundary-matching network for temporal action proposal generation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 3889\u20133898","DOI":"10.1109\/ICCV.2019.00399"},{"key":"6138_CR77","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2023.103692","volume":"232","author":"M Yang","year":"2023","unstructured":"Yang M, Chen G, Zheng Y-D, Lu T, Wang L (2023) Basictad: an astounding rgb-only baseline for temporal action detection. Comput Vis Image Underst 232:103692","journal-title":"Comput Vis Image Underst"},{"key":"6138_CR78","doi-asserted-by":"crossref","unstructured":"Yang H, Wu W, Wang L, Jin S, Xia B, Yao H, Huang H (2022) Temporal action proposal generation with background constraint. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol 36, pp 3054\u20133062","DOI":"10.1609\/aaai.v36i3.20212"},{"key":"6138_CR79","doi-asserted-by":"crossref","unstructured":"Shi D, Zhong Y, Cao Q, Zhang J, Ma L, Li J, Tao D (2022) React: temporal action detection with relational queries. In: European Conference on Computer Vision. Springer, pp 105\u2013121","DOI":"10.1007\/978-3-031-20080-9_7"},{"key":"6138_CR80","doi-asserted-by":"crossref","unstructured":"Cheng F, Bertasius G (2022) Tallformer: temporal action localization with a long-memory transformer. In: European Conference on Computer Vision. Springer, pp 503\u2013521","DOI":"10.1007\/978-3-031-19830-4_29"},{"key":"6138_CR81","doi-asserted-by":"crossref","unstructured":"Weng Y, Pan Z, Han M, Chang X, Zhuang B (2022) An efficient spatio-temporal pyramid transformer for action detection. In: European Conference on Computer Vision. Springer, pp 358\u2013375","DOI":"10.1007\/978-3-031-19830-4_21"},{"key":"6138_CR82","doi-asserted-by":"crossref","unstructured":"Zeng R, Huang W, Tan M, Rong Y, Zhao P, Huang J, Gan C (2019) Graph convolutional networks for temporal action localization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 7094\u20137103","DOI":"10.1109\/ICCV.2019.00719"},{"key":"6138_CR83","doi-asserted-by":"crossref","unstructured":"Carreira J, Zisserman A (2017) Quo vadis, action recognition? A new model and the kinetics dataset. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 6299\u20136308","DOI":"10.1109\/CVPR.2017.502"},{"key":"6138_CR84","doi-asserted-by":"crossref","unstructured":"Tran D, Wang H, Torresani L, Ray J, LeCun Y, Paluri M (2018) A closer look at spatiotemporal convolutions for action recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 6450\u20136459","DOI":"10.1109\/CVPR.2018.00675"},{"key":"6138_CR85","doi-asserted-by":"crossref","unstructured":"Alwassel H, Giancola S, Ghanem B (2021) Tsp: temporally-sensitive pretraining of video encoders for localization tasks. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 3173\u20133183","DOI":"10.1109\/ICCVW54120.2021.00356"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-024-06138-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-024-06138-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-024-06138-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,8]],"date-time":"2024-07-08T12:24:10Z","timestamp":1720441450000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-024-06138-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,6]]},"references-count":85,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2024,8]]}},"alternative-id":["6138"],"URL":"https:\/\/doi.org\/10.1007\/s11227-024-06138-1","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,5,6]]},"assertion":[{"value":"9 April 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 May 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"All the authors do not have any possible conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}},{"value":"Not applicable.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}}]}}