{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T05:27:36Z","timestamp":1781587656079,"version":"3.54.5"},"reference-count":53,"publisher":"Tech Science Press","issue":"2","license":[{"start":{"date-parts":[[2025,2,23]],"date-time":"2025-02-23T00:00:00Z","timestamp":1740268800000},"content-version":"vor","delay-in-days":53,"URL":"https:\/\/doi.org\/10.32604\/TSP-CROSSMARKPOLICY"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["CMC"],"published-print":{"date-parts":[[2025]]},"DOI":"10.32604\/cmc.2024.057392","type":"journal-article","created":{"date-parts":[[2024,12,20]],"date-time":"2024-12-20T02:44:18Z","timestamp":1734662658000},"page":"2389-2414","update-policy":"https:\/\/doi.org\/10.32604\/tsp-crossmarkpolicy","source":"Crossref","is-referenced-by-count":1,"title":["ACSF-ED: Adaptive Cross-Scale Fusion Encoder-Decoder for Spatio-Temporal Action Detection"],"prefix":"10.32604","volume":"82","author":[{"given":"Wenju","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bang","family":"Tang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zehua","family":"Gu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sen","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jianfei","family":"Hao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"17807","published-online":{"date-parts":[[2025]]},"reference":[{"key":"ref1","series-title":"32nd IEEE Int. Conf. Robot Human Interact. Commun., RO-MAN 2023","first-page":"884","article-title":"Abnormal detection of worker by interaction analysis of accident-causing objects","author":"Kim","year":"Aug. 28\u201331, 2023"},{"key":"ref2","series-title":"2021 Int. Conf. Artif. Intell., Virtual Real. Visual., AIVRV 2021","article-title":"Action recognition system for security monitoring","volume":"12153","author":"Yang","year":"Nov. 19\u201321, 2021"},{"key":"ref3","series-title":"2021 Int. Conf. Digital Image Comput.: Tech. Appl., DICTA 2021","article-title":"Improved spatio-temporal action localization for surveillance videos","author":"Liang","year":"Nov. 29\u2013Dec. 1, 2021, pp. 1\u20138"},{"key":"ref4","series-title":"2nd Int. Conf. Comput. Adv. ICCA 2022","first-page":"367","article-title":"A constructive review on pedestrian action detection, recognition and prediction","author":"Khan","year":"Mar. 10\u201312, 2022"},{"key":"ref5","series-title":"2023 Int. Joint Conf. Neural Netw., IJCNN 2023","article-title":"Video-based driver action recognition via spatial-temporal and motion deep learning","author":"Ma","year":"Jun. 18\u201323, 2023, pp. 1\u20139."},{"key":"ref6","series-title":"17th IEEE\/CVF Int. Conf. Comput. Vis., ICCV 2019","first-page":"6201","article-title":"Slowfast networks for video recognition","author":"Feichtenhofer","year":"Oct. 27\u2013Nov. 2, 2019"},{"key":"ref7","series-title":"2022 IEEE\/CVF Conf. Comput. Vis. Pattern Recognit., CVPR 2022","first-page":"13588","article-title":"TubeR: Tubelet transformer for video action detection","author":"Zhao","year":"Jun. 19\u201324, 2022"},{"key":"ref8","unstructured":"O. Kopuklu, X. Wei, and G. Rigoll, \u201cYou only watch once: A unified CNN architecture for real-time spatiotemporal action localization,\u201d 2019, arXiv:1911.06644."},{"key":"ref9","series-title":"16th IEEE Int. Conf. Comput. Vis., ICCV 2017","first-page":"5823","article-title":"Tube convolutional neural network (T-CNN) for action detection in videos","author":"Hou","year":"Oct. 22\u201329, 2017"},{"key":"ref10","series-title":"16th IEEE Int. Conf. Comput. Vis., ICCV 2017","first-page":"4415","article-title":"Action tubelet detector for spatio-temporal action localization","author":"Kalogeiton","year":"Oct. 22\u201329, 2017"},{"key":"ref11","series-title":"14th European Conf. Comput. Vis., ECCV 2016","first-page":"21","article-title":"SSD: Single shot multibox detector","author":"Liu","year":"Oct. 8\u201316, 2016"},{"key":"ref12","series-title":"32nd IEEE\/CVF Conf. Comput. Vis. Pattern Recognit., CVPR 2019","first-page":"264","article-title":"STEP: Spatio-temporal progressive learning for video action detection","author":"Yang","year":"Jun. 16\u201320, 2019"},{"key":"ref13","series-title":"16th Eur. Conf. Comput. Vis., ECCV 2020","first-page":"68","article-title":"Actions as moving points","author":"Li","year":"Aug. 23\u201328, 2020"},{"key":"ref14","series-title":"2021 Int. Jt. Conf. Neural Netw., IJCNN 2021","first-page":"1","article-title":"Spatio-temporal action detector with self-attention","author":"Ma","year":"Jul. 18\u201322, 2021"},{"key":"ref15","series-title":"32nd Conf. Neural Inf. Process. Syst., NeurIPS 2018","first-page":"7610","article-title":"VideocapsuleNet: A simplified network for action detection","author":"Duarte","year":"Dec. 2\u20138, 2018"},{"key":"ref16","series-title":"32nd IEEE\/CVF Conf. Comput. Vis. Pattern Recognit., CVPR 2019","first-page":"11979","article-title":"TACNet: Transition-aware context network for spatio-temporal action detection","author":"Song","year":"Jun. 16\u201320, 2019"},{"key":"ref17","series-title":"IEEE Conf. Comput. Vis. Pattern Recognit., CVPR 2015","first-page":"759","article-title":"Finding action tubes","author":"Gkioxari","year":"Jun. 7\u201312, 2015"},{"key":"ref18","series-title":"27th British Mach. Vis. Conf., BMVC 2016","first-page":"58.1","article-title":"Deep learning for detecting multiple space-time action tubes in videos","author":"Saha","year":"Sep. 19\u201322, 2016"},{"key":"ref19","series-title":"21st ACM Conf. Comput. Commun. Secur., CCS 2014","first-page":"744","article-title":"Multi-region two-stream R-CNN for action detection","author":"Peng","year":"Nov. 3\u20137, 2014"},{"key":"ref20","series-title":"28th British Mach. Vis. Conf., BMVC 2017","article-title":"Spatio-temporal action detection with cascade proposal and location anticipation","author":"Yang","year":"Sep. 4\u20137, 2017"},{"key":"ref21","series-title":"31st Meet. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit., CVPR 2018","first-page":"6047","article-title":"AVA: A video dataset of spatio-temporally localized atomic visual actions","author":"Gu","year":"Jun. 18\u201322, 2018"},{"key":"ref22","series-title":"2020 IEEE\/CVF Conf. Comput. Vis. Pattern Recognit., CVPR 2020","first-page":"200","article-title":"Expanding architectures for efficient video recognition","author":"Feichtenhofer","year":"Jun. 14\u201319, 2020"},{"key":"ref23","series-title":"18th IEEE\/CVF Int. Conf. Comput. Vis., ICCV 2021","first-page":"8158","article-title":"Watch only once: An end-to-end video action detection framework","author":"Chen","year":"Oct. 11\u2013Oct. 17, 2021"},{"key":"ref24","doi-asserted-by":"crossref","unstructured":"J. Yang and K. Dai, \u201cYOWOv2: A stronger yet efficient multi-level detection framework for real-time spatio-temporal action detection,\u201d 2023, arXiv:2302.06848.","DOI":"10.2139\/ssrn.4485402"},{"key":"ref25","unstructured":"N. D. D. Manh, D. V. Hang, J. C. Wang, and B. D. Nhan, \u201cYOWOv3: An efficient and generalized framework for human action detection and recognition,\u201d 2024, arXiv:2408.02623."},{"key":"ref26","series-title":"IEEE Conf. Comput. Vis. Pattern Recognit., CVPR 2015","first-page":"1","article-title":"Going deeper with convolutions","author":"Szegedy","year":"Jun. 7\u201312, 2015"},{"key":"ref27","doi-asserted-by":"crossref","first-page":"1904","DOI":"10.1109\/TPAMI.2015.2389824","article-title":"Spatial pyramid pooling in deep convolutional networks for visual recognition","volume":"37","author":"He","year":"2015","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"ref28","series-title":"30th IEEE Conf. Comput. Vis. Pattern Recognit., CVPR 2017","first-page":"6230","article-title":"Pyramid scene parsing network","author":"Zhao","year":"Jul. 21\u201326, 2017"},{"key":"ref29","series-title":"30th IEEE Conf. Comput. Vis. Pattern Recognit., CVPR 2017","first-page":"936","article-title":"Feature pyramid networks for object detection","author":"Lin","year":"Jul. 21\u201326, 2017"},{"key":"ref30","unstructured":"S. Liu, D. Huang, and Y. Wang, \u201cLearning spatial fusion for single-shot object detection,\u201d 2019, arXiv:1911.09516."},{"key":"ref31","series-title":"27th IEEE Conf. Comput. Vis. Pattern Recognit., CVPR 2014","first-page":"580","article-title":"Rich feature hierarchies for accurate object detection and semantic segmentation","author":"Girshick","year":"Jun. 23\u201328, 2014"},{"key":"ref32","series-title":"17th IEEE\/CVF Int. Conf. Comput. Vis., ICCV 2019","first-page":"9626","article-title":"FCOS: Fully convolutional one-stage object detection","author":"Tian","year":"Oct. 27\u2013Nov. 2, 2019"},{"key":"ref33","unstructured":"J. Redmon and A. Farhadi, \u201cYOLOv3: An incremental improvement,\u201d 2018, arXiv:1804.02767."},{"key":"ref34","unstructured":"A. Bochkovskiy, C. -Y. Wang, and H. -Y. M. Liao, \u201cYOLOv4: Optimal speed and accuracy of object detection,\u201d 2020, arXiv:2004.10934."},{"key":"ref35","series-title":"2023 IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"7464","article-title":"YOLOv7: Trainable bag-of-freebies sets new state-of-the-art for real-time object detectors","author":"Wang","year":"2023"},{"key":"ref36","unstructured":"C. Li et al., \u201cYOLOv6 v3.0: A full-scale reloading,\u201d 2023, arXiv:2301.05586."},{"key":"ref37","unstructured":"Z. Ge, S. Liu, F. Wang, Z. Li, and J. Sun, \u201cYOLOX: Exceeding YOLO series in 2021,\u201d 2021, arXiv:2107.08430."},{"key":"ref38","series-title":"31st Annual Conf. Neural Inf. Process. Syst., NIPS 2017","first-page":"5999","article-title":"Attention is all you need","author":"Vaswani","year":"Dec. 4\u20139, 2017"},{"key":"ref39","series-title":"Comput. Vis.-ECCV 2020: 16th Eur. Conf.","first-page":"213","article-title":"End-to-end object detection with transformers","author":"Carion","year":"2020"},{"key":"ref40","series-title":"10th Int. Conf. Learn. Representations, ICLR 2022","article-title":"Dynamic anchor boxes are better queries for detr","author":"Liu","year":"Apr. 25\u201329, 2022"},{"key":"ref41","doi-asserted-by":"crossref","first-page":"2239","DOI":"10.1109\/TPAMI.2023.3335410","article-title":"DN-DETR: Accelerate DETR training by introducing query DeNoising","volume":"46","author":"Li","year":"2024","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"ref42","series-title":"2024 IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"16965","article-title":"DETRs beat YOLOs on real-time object detection","author":"Zhao","year":"2024"},{"key":"ref43","series-title":"2nd Int. Conf. Innov. Technol., INOCON 2023","article-title":"Deep-learning residual network based image analysis for an efficient two-stage recognition of neurological disorders","author":"Battula","year":"Mar. 3\u20135, 2023"},{"key":"ref44","unstructured":"J. Lin, X. Mao, Y. Chen, L. Xu, Y. He and H. Xue, \u201cD2ETR: Decoder-only DETR with computationally efficient cross-scale attention,\u201d 2022, arXiv:2203.00860."},{"key":"ref45","series-title":"30th IEEE Conf. Comput. Vis. Pattern Recognit., CVPR 2017","first-page":"5987","article-title":"Aggregated residual transformations for deep neural networks","author":"Xie","year":"Jul. 21\u201326, 2017"},{"key":"ref46","series-title":"2021 IEEE\/CVF Conf. Comput. Vis. Pattern Recognit., CVPR 2021","first-page":"13728","article-title":"RepVGG: Making VGG-style ConvNets great again","author":"Ding","year":"Jun. 19\u201325, 2021"},{"key":"ref47","series-title":"32nd IEEE\/CVF Conf. Comput. Vis. Pattern Recognit., CVPR 2019","first-page":"3141","article-title":"Dual attention network for scene segmentation","author":"Fu","year":"Jun. 16\u201320, 2019"},{"key":"ref48","series-title":"11th Int. Conf. Learn. Representations, ICLR 2023","article-title":"DINO: DETR with improved denoising anchor boxes for end-to-end object detection","year":"May 1\u20135, 2023"},{"key":"ref49","series-title":"2022 IEEE Int. Conf. Multimed. Expo, ICME 2022","article-title":"CAT: Cross attention in vision transformer","author":"Lin","year":"Jul. 18\u201322, 2022"},{"key":"ref50","unstructured":"K. Soomro, A. Zamir, and M. J. A. Shah, \u201cUCF101: A dataset of 101 human actions classes from videos in the wild,\u201d 2012, arXiv:1212.0402."},{"key":"ref51","series-title":"2013 14th IEEE Int. Conf. Comput. Vis., ICCV 2013","first-page":"3192","article-title":"Towards understanding action recognition","author":"Jhuang","year":"Dec. 1\u20138, 2013"},{"key":"ref52","doi-asserted-by":"crossref","first-page":"303","DOI":"10.1007\/s11263-009-0275-4","article-title":"The pascal visual object classes (VOC) challenge","volume":"88","author":"Everingham","year":"2010","journal-title":"Int. J. Comput. Vis."},{"key":"ref53","doi-asserted-by":"crossref","first-page":"104","DOI":"10.1109\/TCSVT.2018.2887283","article-title":"CNN-based multiple path search for action tube detection in videos","volume":"30","author":"Alwando","year":"2020","journal-title":"IEEE Trans. Circuits Syst. Video Technol."}],"container-title":["Computers, Materials &amp; Continua"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/cdn.techscience.cn\/files\/cmc\/2025\/TSP_CMC-82-2\/TSP_CMC_57392\/TSP_CMC_57392.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,14]],"date-time":"2025-11-14T06:08:44Z","timestamp":1763100524000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.techscience.com\/cmc\/v82n2\/59445"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"references-count":53,"journal-issue":{"issue":"2","published-online":{"date-parts":[[2025]]},"published-print":{"date-parts":[[2025]]}},"URL":"https:\/\/doi.org\/10.32604\/cmc.2024.057392","relation":{},"ISSN":["1546-2226"],"issn-type":[{"value":"1546-2226","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"2024-08-16","order":0,"name":"received","label":"Received","group":{"name":"publication_history","label":"Publication History"}},{"value":"2024-11-06","order":1,"name":"accepted","label":"Accepted","group":{"name":"publication_history","label":"Publication History"}},{"value":"2025-02-17","order":2,"name":"published","label":"Published Online","group":{"name":"publication_history","label":"Publication History"}}]}}