{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T17:34:12Z","timestamp":1772818452228,"version":"3.50.1"},"reference-count":46,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2022,11,16]],"date-time":"2022-11-16T00:00:00Z","timestamp":1668556800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,11,16]],"date-time":"2022-11-16T00:00:00Z","timestamp":1668556800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int. J. Mach. Learn. &amp; Cyber."],"published-print":{"date-parts":[[2023,5]]},"DOI":"10.1007\/s13042-022-01720-6","type":"journal-article","created":{"date-parts":[[2022,11,16]],"date-time":"2022-11-16T17:04:57Z","timestamp":1668618297000},"page":"1683-1693","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["GMNet: an action recognition network with global motion representation"],"prefix":"10.1007","volume":"14","author":[{"given":"Mingwei","family":"Liu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yi","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,11,16]]},"reference":[{"key":"1720_CR1","doi-asserted-by":"crossref","unstructured":"Karpathy A, Toderici G, Shetty S, Leung T, Sukthankar R, Fei-Fei L (2014) Large-scale video classification with convolutional neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1725\u20131732","DOI":"10.1109\/CVPR.2014.223"},{"key":"1720_CR2","doi-asserted-by":"crossref","unstructured":"Wang L, Qiao Y, Tang X (2015) Action recognition with trajectory-pooled deep-convolutional descriptors. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4305\u20134314","DOI":"10.1109\/CVPR.2015.7299059"},{"key":"1720_CR3","doi-asserted-by":"publisher","unstructured":"Tran D, Bourdev L, Fergus R, Torresani L, Paluri M (2015) Learning spatiotemporal features with 3d convolutional networks. In: 2015 IEEE International Conference on Computer Vision (ICCV), pp. 4489\u20134497. https:\/\/doi.org\/10.1109\/ICCV.2015.510","DOI":"10.1109\/ICCV.2015.510"},{"key":"1720_CR4","doi-asserted-by":"crossref","unstructured":"Stroud J, Ross D, Sun C, Deng J, Sukthankar R (2020) D3d: Distilled 3d networks for video action recognition. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 625\u2013634","DOI":"10.1109\/WACV45572.2020.9093274"},{"key":"1720_CR5","doi-asserted-by":"crossref","unstructured":"Carreira J, Zisserman A (2017) Quo vadis, action recognition? a new model and the kinetics dataset. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6299\u20136308","DOI":"10.1109\/CVPR.2017.502"},{"key":"1720_CR6","unstructured":"Simonyan K, Zisserman A (2014) Two-stream convolutional networks for action recognition in videos. Advances in neural information processing systems 27"},{"key":"1720_CR7","doi-asserted-by":"crossref","unstructured":"Wang L, Xiong Y, Wang Z, Qiao Y, Lin D, Tang X, Gool LV (2016) Temporal segment networks: Towards good practices for deep action recognition. In: European Conference on Computer Vision, pp. 20\u201336. Springer","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"1720_CR8","doi-asserted-by":"crossref","unstructured":"Lin J, Gan C, Han S (2019) Tsm: Temporal shift module for efficient video understanding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7083\u20137093","DOI":"10.1109\/ICCV.2019.00718"},{"key":"1720_CR9","doi-asserted-by":"crossref","unstructured":"Zhou B, Andonian A, Oliva A, Torralba A (2018) Temporal relational reasoning in videos. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 803\u2013818","DOI":"10.1007\/978-3-030-01246-5_49"},{"key":"1720_CR10","doi-asserted-by":"crossref","unstructured":"Goyal R, Kahou SE, Michalski V, Materzynska J, Westphal S, Kim H, Haenel V, Fruend I, Yianilos P, Mueller-Freitag M et al. (2017) The\u201c something something\u201d video database for learning and evaluating visual common sense. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 5842\u20135850","DOI":"10.1109\/ICCV.2017.622"},{"key":"1720_CR11","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"issue":"3","key":"1720_CR12","doi-asserted-by":"publisher","first-page":"823","DOI":"10.1007\/s13042-020-01204-5","volume":"12","author":"D Zhuang","year":"2021","unstructured":"Zhuang D, Jiang M, Kong J, Liu T (2021) Spatiotemporal attention enhanced features fusion network for action recognition. Int J Mach Learn Cybern 12(3):823\u2013841","journal-title":"Int J Mach Learn Cybern"},{"key":"1720_CR13","doi-asserted-by":"crossref","unstructured":"Donahue J, Hendricks LA, Guadarrama S, Rohrbach M, Venugopalan S, Saenko K, Darrell T (2015) Long-term recurrent convolutional networks for visual recognition and description. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2625\u20132634","DOI":"10.1109\/CVPR.2015.7298878"},{"key":"1720_CR14","doi-asserted-by":"crossref","unstructured":"Kwon H, Kim M, Kwak S, Cho M (2020) Motionsqueeze: Neural motion feature learning for video understanding. In: European Conference on Computer Vision, pp. 345\u2013362. Springer","DOI":"10.1007\/978-3-030-58517-4_21"},{"key":"1720_CR15","doi-asserted-by":"crossref","unstructured":"Feichtenhofer C, Fan H, Malik J, He K (2019) Slowfast networks for video recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6202\u20136211","DOI":"10.1109\/ICCV.2019.00630"},{"key":"1720_CR16","doi-asserted-by":"crossref","unstructured":"Fan L, Huang W, Gan C, Ermon S, Gong B, Huang J (2018) End-to-end learning of motion representation for video understanding. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6016\u20136025","DOI":"10.1109\/CVPR.2018.00630"},{"key":"1720_CR17","doi-asserted-by":"crossref","unstructured":"Piergiovanni A, Ryoo MS (2019) Representation flow for action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9945\u20139953","DOI":"10.1109\/CVPR.2019.01018"},{"key":"1720_CR18","doi-asserted-by":"crossref","unstructured":"Jiang B, Wang M, Gan W, Wu W, Yan J (2019) Stm: Spatiotemporal and motion encoding for action recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2000\u20132009","DOI":"10.1109\/ICCV.2019.00209"},{"key":"1720_CR19","doi-asserted-by":"crossref","unstructured":"Lee M, Lee S, Son S, Park G, Kwak N (2018) Motion feature network: Fixed motion filter for action recognition. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 387\u2013403","DOI":"10.1007\/978-3-030-01249-6_24"},{"key":"1720_CR20","doi-asserted-by":"crossref","unstructured":"Sun S, Kuang Z, Sheng L, Ouyang W, Zhang W (2018) Optical flow guided feature: A fast and robust motion representation for video action recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1390\u20131399","DOI":"10.1109\/CVPR.2018.00151"},{"key":"1720_CR21","doi-asserted-by":"crossref","unstructured":"Wang H, Tran D, Torresani L, Feiszli M (2020) Video modeling with correlation networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 352\u2013361","DOI":"10.1109\/CVPR42600.2020.00043"},{"issue":"6","key":"1720_CR22","doi-asserted-by":"publisher","first-page":"2799","DOI":"10.1109\/TIP.2018.2890749","volume":"28","author":"Z Tu","year":"2019","unstructured":"Tu Z, Li H, Zhang D, Dauwels J, Li B, Yuan J (2019) Action-stage emphasized spatiotemporal VLAD for video action recognition. IEEE Trans Image Process 28(6):2799\u20132812","journal-title":"IEEE Trans Image Process"},{"key":"1720_CR23","doi-asserted-by":"crossref","unstructured":"Dosovitskiy A, Fischer P, Ilg E, Hausser P, Hazirbas C, Golkov V, Van Der\u00a0Smagt P, Cremers D, Brox T (2015) Flownet: Learning optical flow with convolutional networks. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2758\u20132766","DOI":"10.1109\/ICCV.2015.316"},{"key":"1720_CR24","doi-asserted-by":"crossref","unstructured":"Sun D, Yang X, Liu M-Y, Kautz J (2018) Pwc-net: Cnns for optical flow using pyramid, warping, and cost volume. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 8934\u20138943","DOI":"10.1109\/CVPR.2018.00931"},{"key":"1720_CR25","doi-asserted-by":"crossref","unstructured":"Teed Z, Deng J (2020) Raft: Recurrent all-pairs field transforms for optical flow. In: European Conference on Computer Vision, pp. 402\u2013419. Springer","DOI":"10.1007\/978-3-030-58536-5_24"},{"key":"1720_CR26","doi-asserted-by":"crossref","unstructured":"Honari S, Molchanov P, Tyree S, Vincent P, Pal C, Kautz J (2018) Improving landmark localization with semi-supervised learning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1546\u20131555","DOI":"10.1109\/CVPR.2018.00167"},{"key":"1720_CR27","doi-asserted-by":"crossref","unstructured":"Lee J, Kim D, Ponce J, Ham B (2019) Sfnet: Learning object-aware semantic correspondence. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2278\u20132287","DOI":"10.1109\/CVPR.2019.00238"},{"key":"1720_CR28","doi-asserted-by":"crossref","unstructured":"Qin Z, Zhang P, Wu F, Li X (2021) Fcanet: Frequency channel attention networks. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 783\u2013792","DOI":"10.1109\/ICCV48922.2021.00082"},{"key":"1720_CR29","doi-asserted-by":"crossref","unstructured":"Liu Z, Luo D, Wang Y, Wang L, Tai Y, Wang C, Li J, Huang F, Lu T (2020) Teinet: Towards an efficient architecture for video recognition. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, pp. 11669\u201311676","DOI":"10.1609\/aaai.v34i07.6836"},{"key":"1720_CR30","doi-asserted-by":"crossref","unstructured":"Li Y, Ji B, Shi X, Zhang J, Kang B, Wang L (2020) Tea: Temporal excitation and aggregation for action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 909\u2013918","DOI":"10.1109\/CVPR42600.2020.00099"},{"key":"1720_CR31","doi-asserted-by":"crossref","unstructured":"Liu Z, Wang L, Wu W, Qian C, Lu T (2021) Tam: Temporal adaptive module for video recognition, 13708\u201313718","DOI":"10.1109\/ICCV48922.2021.01345"},{"key":"1720_CR32","doi-asserted-by":"crossref","unstructured":"Zolfaghari M, Singh K, Brox T (2018) Eco: Efficient convolutional network for online video understanding. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 695\u2013712","DOI":"10.1007\/978-3-030-01216-8_43"},{"key":"1720_CR33","doi-asserted-by":"crossref","unstructured":"Wang X, Gupta A (2018) Videos as space-time region graphs. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 399\u2013417","DOI":"10.1007\/978-3-030-01228-1_25"},{"key":"1720_CR34","unstructured":"Bertasius G, Wang H, Torresani L (2021) Is space-time attention all you need for video understanding? 2(3), 4"},{"key":"1720_CR35","doi-asserted-by":"crossref","unstructured":"Li X, Liu C, Shuai B, Zhu Y, Chen H, Tighe J (2022) Nuta: Non-uniform temporal aggregation for action recognition, 3683\u20133692","DOI":"10.1109\/WACV51458.2022.00090"},{"key":"1720_CR36","doi-asserted-by":"crossref","unstructured":"Li X, Wang Y, Zhou Z, Qiao Y (2020) Smallbignet: Integrating core and contextual views for video classification. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1092\u20131101","DOI":"10.1109\/CVPR42600.2020.00117"},{"issue":"9","key":"1720_CR37","doi-asserted-by":"publisher","first-page":"4839","DOI":"10.1109\/TPAMI.2021.3076522","volume":"44","author":"S Kumawat","year":"2022","unstructured":"Kumawat S, Verma M, Nakashima Y, Raman S (2022) Depthwise Spatio-Temporal STFT Convolutional Neural Networks for Human Action Recognition. IEEE Trans Pattern Anal Mach Intell 44(9):4839\u20134851. https:\/\/doi.org\/10.1109\/TPAMI.2021.3076522","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"1720_CR38","doi-asserted-by":"crossref","unstructured":"Wang L, Tong Z, Ji B, Wu G (2021) Tdn: Temporal difference networks for efficient action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1895\u20131904","DOI":"10.1109\/CVPR46437.2021.00193"},{"key":"1720_CR39","doi-asserted-by":"publisher","unstructured":"Wang M, Xing J, Su J, Chen J, Yong L (2022) Learning SpatioTemporal and Motion Features in a Unified 2D Network for Action Recognition. IEEE Transactions on Pattern Analysis and Machine Intelligence, 1\u20131. https:\/\/doi.org\/10.1109\/TPAMI.2022.3173658","DOI":"10.1109\/TPAMI.2022.3173658"},{"key":"1720_CR40","doi-asserted-by":"crossref","unstructured":"Materzynska J, Berger G, Bax I, Memisevic R (2019) The jester dataset: A large-scale video dataset of human gestures. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision Workshops, pp. 0\u20130","DOI":"10.1109\/ICCVW.2019.00349"},{"key":"1720_CR41","doi-asserted-by":"crossref","unstructured":"Li Y, Li Y, Vasconcelos N (2018) Resound: Towards action recognition without representation bias. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 513\u2013528","DOI":"10.1007\/978-3-030-01231-1_32"},{"key":"1720_CR42","doi-asserted-by":"crossref","unstructured":"Wang X, Girshick R, Gupta A, He K (2018) Non-local neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7794\u20137803","DOI":"10.1109\/CVPR.2018.00813"},{"key":"1720_CR43","doi-asserted-by":"crossref","unstructured":"Kanojia G, Kumawat S, Raman S (2019) Attentive spatio-temporal representation learning for diving classification. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops","DOI":"10.1109\/CVPRW.2019.00302"},{"key":"1720_CR44","doi-asserted-by":"crossref","unstructured":"Luo C, Yuille AL (2019) Grouped spatial-temporal aggregation for efficient action recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5512\u20135521","DOI":"10.1109\/ICCV.2019.00561"},{"key":"1720_CR45","doi-asserted-by":"crossref","unstructured":"Liu H, Ren B, Liu M, Ding R (2020) Grouped temporal enhancement module for human action recognition. In: 2020 IEEE International Conference on Image Processing (ICIP), pp. 1801\u20131805 . IEEE","DOI":"10.1109\/ICIP40778.2020.9190958"},{"key":"1720_CR46","doi-asserted-by":"crossref","unstructured":"Selvaraju RR, Cogswell M, Das A, Vedantam R, Parikh D, Batra D (2017) Grad-cam: Visual explanations from deep networks via gradient-based localization. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 618\u2013626","DOI":"10.1109\/ICCV.2017.74"}],"container-title":["International Journal of Machine Learning and Cybernetics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13042-022-01720-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13042-022-01720-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13042-022-01720-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,4,20]],"date-time":"2023-04-20T04:11:36Z","timestamp":1681963896000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13042-022-01720-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,11,16]]},"references-count":46,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2023,5]]}},"alternative-id":["1720"],"URL":"https:\/\/doi.org\/10.1007\/s13042-022-01720-6","relation":{},"ISSN":["1868-8071","1868-808X"],"issn-type":[{"value":"1868-8071","type":"print"},{"value":"1868-808X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,11,16]]},"assertion":[{"value":"15 June 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 November 2022","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 November 2022","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"I declare I have no financial support and personal relationships with other people or organizations that can inappropriate influence our work. There is no professional or other personal interest of any nature or kind in any products, services, or company that could be construed as influencing the position presented in, or the review of, the manuscript entitled.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}