{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T11:54:34Z","timestamp":1784116474123,"version":"3.55.0"},"reference-count":50,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2021,7,13]],"date-time":"2021-07-13T00:00:00Z","timestamp":1626134400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,7,13]],"date-time":"2021-07-13T00:00:00Z","timestamp":1626134400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2022,2]]},"DOI":"10.1007\/s11227-021-03957-4","type":"journal-article","created":{"date-parts":[[2021,7,13]],"date-time":"2021-07-13T11:04:12Z","timestamp":1626174252000},"page":"2873-2908","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":44,"title":["A transfer learning-based efficient spatiotemporal human action recognition framework for long and overlapping action classes"],"prefix":"10.1007","volume":"78","author":[{"given":"Muhammad","family":"Bilal","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Muazzam","family":"Maqsood","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sadaf","family":"Yasmin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Najam Ul","family":"Hasan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Seungmin","family":"Rho","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2021,7,13]]},"reference":[{"key":"3957_CR1","unstructured":"T. YouTube-Team (2020) 60 hours per minute and 4 billion views a day on YouTube. Youtube official Blog. https:\/\/blog.youtube\/news-and-events\/holy-nyans-60-hours-per-minute-and-4\/ (Accessed 12 Nov 2020)"},{"key":"3957_CR2","unstructured":"Wojcicki S (2020) YouTube at 15: my personal journey and the road ahead. Youtube official Blog. https:\/\/blog.youtube\/news-and-events\/youtube-at-15-my-personal-journey (Accessed 12 Nov 2020)"},{"key":"3957_CR3","unstructured":"Cisco (2020) Cisco annual internet report (2018\u20132023) white paper. https:\/\/www.cisco.com\/c\/en\/us\/solutions\/collateral\/executive-perspectives\/annual-internet-report\/white-paper-c11-741490.html (Accessed 12 Nov 2020)"},{"key":"3957_CR4","doi-asserted-by":"crossref","unstructured":"Karpathy A, Toderici G, Shetty S, Leung T, Sukthankar R, Fei-Fei L (2014) Large-scale video classification with convolutional neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 1725\u20131732","DOI":"10.1109\/CVPR.2014.223"},{"key":"3957_CR5","unstructured":"Simonyan K, Zisserman A (2014) Very deep convolutional networks for large-scale image recognition. arXiv preprint http:\/\/arxiv.org\/abs\/arXiv:1409.1556"},{"key":"3957_CR6","doi-asserted-by":"crossref","unstructured":"Donahue J et al (2015) Long-term recurrent convolutional networks for visual recognition and description. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 2625\u20132634","DOI":"10.1109\/CVPR.2015.7298878"},{"key":"3957_CR7","doi-asserted-by":"crossref","unstructured":"Tran D, Bourdev L, Fergus R, Torresani L, Paluri M (2015) Learning spatiotemporal features with 3d convolutional networks. In: Proceedings of the IEEE International Conference on Computer Vision, pp 4489\u20134497","DOI":"10.1109\/ICCV.2015.510"},{"key":"3957_CR8","doi-asserted-by":"crossref","unstructured":"Feichtenhofer C, Pinz A, Zisserman A (2016) Convolutional two-stream network fusion for video action recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 1933\u20131941","DOI":"10.1109\/CVPR.2016.213"},{"key":"3957_CR9","doi-asserted-by":"crossref","unstructured":"Wang L et al (2016) Temporal segment networks: towards good practices for deep action recognition. In: European Conference on Computer Vision, Springer, pp 20\u201336","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"3957_CR10","doi-asserted-by":"crossref","unstructured":"Girdhar R, Ramanan D, Gupta A, Sivic J, Russell B (2017) Actionvlad: learning spatio-temporal aggregation for action classification. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 971\u2013980","DOI":"10.1109\/CVPR.2017.337"},{"key":"3957_CR11","doi-asserted-by":"crossref","unstructured":"Zhu Y, Lan Z, Newsam S, Hauptmann A (2018) Hidden two-stream convolutional networks for action recognition. In: Asian Conference on Computer Vision, Springer, pp 363\u2013378","DOI":"10.1007\/978-3-030-20893-6_23"},{"key":"3957_CR12","unstructured":"Diba A, et al. (2017) Temporal 3d convnets: new architecture and transfer learning for video classification. arXiv preprint http:\/\/arxiv.org\/abs\/arXiv:1711.08200"},{"key":"3957_CR13","unstructured":"Girdhar R, Ramanan D (2017) Attentional pooling for action recognition. In: Advances in Neural Information Processing Systems, pp 34\u201345"},{"key":"3957_CR14","doi-asserted-by":"crossref","unstructured":"Zheng Z, An G, Wu D, Ruan Q (2020) Global and local knowledge-aware attention network for action recognition. In: IEEE Transactions on Neural Networks and Learning Systems","DOI":"10.1109\/TNNLS.2020.2978613"},{"key":"3957_CR15","doi-asserted-by":"crossref","unstructured":"Long X, Gan C, De Melo G, Wu J, Liu X, Wen S (2018) Attention clusters: purely attention based local feature integration for video classification. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 7834\u20137843","DOI":"10.1109\/CVPR.2018.00817"},{"key":"3957_CR16","doi-asserted-by":"crossref","unstructured":"Girdhar R, Carreira J, Doersch C, Zisserman A (2019) Video action transformer network. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 244\u2013253","DOI":"10.1109\/CVPR.2019.00033"},{"key":"3957_CR17","doi-asserted-by":"publisher","first-page":"43243","DOI":"10.1109\/ACCESS.2020.2977856","volume":"8","author":"J Yu","year":"2020","unstructured":"Yu J et al (2020) A discriminative deep model with feature fusion and temporal attention for human action recognition. IEEE Access 8:43243\u201343255","journal-title":"IEEE Access"},{"key":"3957_CR18","doi-asserted-by":"publisher","first-page":"39920","DOI":"10.1109\/ACCESS.2020.2976496","volume":"8","author":"C Liang","year":"2020","unstructured":"Liang C, Liu D, Qi L, Guan L (2020) Multi-modal human action recognition with sub-action exploiting and class-privacy preserved collaborative representation learning. IEEE Access 8:39920\u201339933","journal-title":"IEEE Access"},{"key":"3957_CR19","doi-asserted-by":"crossref","unstructured":"Nazir S, Qian Y, Yousaf M, Velastin Carroza SA, Izquierdo E, Vazquez E (2019) Human action recognition using multi-kernel learning for temporal residual network","DOI":"10.5220\/0007371104200426"},{"key":"3957_CR20","doi-asserted-by":"crossref","unstructured":"Zhang D, Dai X, Wang Y-F (2020) METAL: minimum effort temporal activity localization in untrimmed videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 3882\u20133892","DOI":"10.1109\/CVPR42600.2020.00394"},{"key":"3957_CR21","doi-asserted-by":"crossref","unstructured":"Yang X, Yang X, Liu M-Y, Xiao F, Davis LS, Kautz J (2019) Step: spatio-temporal progressive learning for video action detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 264\u2013272","DOI":"10.1109\/CVPR.2019.00035"},{"issue":"1","key":"3957_CR22","doi-asserted-by":"publisher","first-page":"221","DOI":"10.1109\/TPAMI.2012.59","volume":"35","author":"S Ji","year":"2012","unstructured":"Ji S, Xu W, Yang M, Yu K (2012) 3D convolutional neural networks for human action recognition. IEEE Trans Pattern Anal Mach Intell 35(1):221\u2013231","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"3957_CR23","doi-asserted-by":"publisher","first-page":"334","DOI":"10.3389\/fnhum.2017.00334","volume":"11","author":"Y-P Lin","year":"2017","unstructured":"Lin Y-P, Jung T-P (2017) Improving EEG-based emotion classification using conditional transfer learning. Front Hum Neurosci 11:334","journal-title":"Front Hum Neurosci"},{"key":"3957_CR24","unstructured":"Zhuang F et al (2019) A comprehensive survey on transfer learning. arXiv preprint http:\/\/arxiv.org\/abs\/arXiv:1911.02685"},{"key":"3957_CR25","doi-asserted-by":"publisher","first-page":"370","DOI":"10.1016\/j.patrec.2018.08.003","volume":"130","author":"K Muhammad","year":"2020","unstructured":"Muhammad K, Hussain T, Baik SW (2020) Efficient CNN based summarization of surveillance videos for resource-constrained devices. Pattern Recogn Lett 130:370\u2013375","journal-title":"Pattern Recogn Lett"},{"key":"3957_CR26","doi-asserted-by":"publisher","first-page":"18174","DOI":"10.1109\/ACCESS.2018.2812835","volume":"6","author":"K Muhammad","year":"2018","unstructured":"Muhammad K, Ahmad J, Mehmood I, Rho S, Baik SW (2018) Convolutional neural networks based fire detection in surveillance videos. IEEE Access 6:18174\u201318183","journal-title":"IEEE Access"},{"key":"3957_CR27","doi-asserted-by":"crossref","unstructured":"Wang C, Yang H, Bartz C, Meinel C (2016) Image captioning with deep bidirectional LSTMs. In: Proceedings of the 24th ACM international conference on Multimedia, pp 988\u2013997","DOI":"10.1145\/2964284.2964299"},{"issue":"2","key":"3957_CR28","first-page":"1","volume":"14","author":"C Wang","year":"2018","unstructured":"Wang C, Yang H, Meinel C (2018) Image captioning with deep bidirectional LSTMs and multi-task learning. ACM Trans Multimedia Comput Commun Appl (TOMM) 14(2):1\u201320","journal-title":"ACM Trans Multimedia Comput Commun Appl (TOMM)"},{"key":"3957_CR29","doi-asserted-by":"crossref","unstructured":"Deng J, Dong W, Socher R, Li L-J, Li K, Fei-Fei L (2009) Imagenet: a large-scale hierarchical image database. In: 2009 IEEE Conference on Computer Vision and Pattern Recognition, pp 248\u2013255","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"3957_CR30","doi-asserted-by":"crossref","unstructured":"O\u2019Mahony N, et al. (2019) Deep learning vs. traditional computer vision. In: Science and Information Conference, Springer, pp 128\u2013144","DOI":"10.1007\/978-3-030-17795-9_10"},{"key":"3957_CR31","first-page":"1","volume":"9","author":"T Georgiou","year":"2019","unstructured":"Georgiou T, Liu Y, Chen W, Lew M (2019) A survey of traditional and deep learning-based feature descriptors for high dimensional data in computer vision. Int J Multimedia Inf Retr 9:1\u201336","journal-title":"Int J Multimedia Inf Retr"},{"key":"3957_CR32","doi-asserted-by":"crossref","unstructured":"Salau AO, Jain S (2019) Feature extraction: a survey of the types, techniques, applications. In: 2019 International Conference on Signal Processing and Communication (ICSC), IEEE, pp 158\u2013164","DOI":"10.1109\/ICSC45622.2019.8938371"},{"issue":"02","key":"3957_CR33","first-page":"73","volume":"1","author":"A Bashar","year":"2019","unstructured":"Bashar A (2019) Survey on evolving deep learning neural network architectures. J Artif Intell 1(02):73\u201382","journal-title":"J Artif Intell"},{"key":"3957_CR34","doi-asserted-by":"crossref","unstructured":"Karthikayani K, Arunachalam A (2020) A survey on deep learning feature extraction techniques. In: AIP Conference Proceedings, vol 2282, no 1, AIP Publishing LLC, p 020035","DOI":"10.1063\/5.0028564"},{"key":"3957_CR35","doi-asserted-by":"crossref","unstructured":"Huang G, Liu Z, Van Der Maaten L, Weinberger KQ (2017) Densely connected convolutional networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 4700\u20134708","DOI":"10.1109\/CVPR.2017.243"},{"key":"3957_CR36","doi-asserted-by":"crossref","unstructured":"Szegedy C, Vanhoucke V, Ioffe S, Shlens J, Wojna Z (2016) Rethinking the inception architecture for computer vision. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 2818\u20132826","DOI":"10.1109\/CVPR.2016.308"},{"key":"3957_CR37","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Identity mappings in deep residual networks. In: European Conference on Computer Vision, Springer, pp 630\u2013645","DOI":"10.1007\/978-3-319-46493-0_38"},{"key":"3957_CR38","doi-asserted-by":"crossref","unstructured":"Chollet F (2017) Xception: deep learning with depthwise separable convolutions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 1251\u20131258","DOI":"10.1109\/CVPR.2017.195"},{"key":"3957_CR39","unstructured":"Soomro K, Zamir AR, Shah M (2012) UCF101: a dataset of 101 human actions classes from videos in the wild. arXiv preprint http:\/\/arxiv.org\/abs\/arXiv:1212.0402"},{"key":"3957_CR40","doi-asserted-by":"crossref","unstructured":"Caetano CA, De Melo VHC, dos Santos JA, Schwartz WR (2017) Activity recognition based on a magnitude-orientation stream network. In: 2017 30th SIBGRAPI Conference on Graphics, Patterns and Images (SIBGRAPI), IEEE, pp 47\u201354","DOI":"10.1109\/SIBGRAPI.2017.13"},{"key":"3957_CR41","doi-asserted-by":"crossref","unstructured":"Dalal N, Triggs B, Schmid C (2006) Human detection using oriented histograms of flow and appearance. In: European Conference on Computer Vision, Springer, pp 428\u2013441","DOI":"10.1007\/11744047_33"},{"key":"3957_CR42","doi-asserted-by":"crossref","unstructured":"Shi F, Laganiere R, Petriu E (2015) Gradient boundary histograms for action recognition. In: 2015 IEEE Winter Conference on Applications of Computer Vision, pp 1107\u20131114","DOI":"10.1109\/WACV.2015.152"},{"key":"3957_CR43","doi-asserted-by":"publisher","first-page":"109","DOI":"10.1016\/j.cviu.2016.03.013","volume":"150","author":"X Peng","year":"2016","unstructured":"Peng X, Wang L, Wang X, Qiao Y (2016) Bag of visual words and fusion methods for action recognition: comprehensive study and good practice. Comput Vis Image Underst 150:109\u2013125","journal-title":"Comput Vis Image Underst"},{"key":"3957_CR44","doi-asserted-by":"crossref","unstructured":"Cai Z, Wang L, Peng X, Qiao Y (2014) Multi-view super vector for action recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 596\u2013603","DOI":"10.1109\/CVPR.2014.83"},{"key":"3957_CR45","unstructured":"Srivastava N, Mansimov E, Salakhudinov R (2015) Unsupervised learning of video representations using lstms. In: International Conference on Machine Learning, pp 843\u2013852"},{"issue":"1","key":"3957_CR46","doi-asserted-by":"publisher","first-page":"102","DOI":"10.1109\/TPAMI.2016.2537337","volume":"39","author":"A-A Liu","year":"2016","unstructured":"Liu A-A, Su Y-T, Nie W-Z, Kankanhalli M (2016) Hierarchical clustering multi-task learning for joint human action grouping and recognition. IEEE Trans Pattern Anal Mach Intell 39(1):102\u2013114","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"3957_CR47","doi-asserted-by":"crossref","unstructured":"Zhu W, Hu J, Sun G, Cao X, Qiao Y (2016) A key volume mining deep framework for action recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 1991\u20131999","DOI":"10.1109\/CVPR.2016.219"},{"key":"3957_CR48","doi-asserted-by":"crossref","unstructured":"Sun L, Jia K, Yeung D-Y, Shi BE (2015) Human action recognition using factorized spatio-temporal convolutional networks. In: Proceedings of the IEEE International Conference on Computer Vision, pp 4597\u20134605","DOI":"10.1109\/ICCV.2015.522"},{"key":"3957_CR49","unstructured":"Simonyan K, Zisserman A (2014) Two-stream convolutional networks for action recognition in videos. In: Advances in neural information processing systems, pp 568\u2013576"},{"key":"3957_CR50","doi-asserted-by":"publisher","first-page":"386","DOI":"10.1016\/j.future.2019.01.029","volume":"96","author":"A Ullah","year":"2019","unstructured":"Ullah A, Muhammad K, Haq IU, Baik SW (2019) Action recognition using optimized deep autoencoder and CNN for surveillance data streams of non-stationary environments. Futur Gener Comput Syst 96:386\u2013397","journal-title":"Futur Gener Comput Syst"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-021-03957-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-021-03957-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-021-03957-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,1,24]],"date-time":"2022-01-24T11:35:19Z","timestamp":1643024119000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-021-03957-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,7,13]]},"references-count":50,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2022,2]]}},"alternative-id":["3957"],"URL":"https:\/\/doi.org\/10.1007\/s11227-021-03957-4","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,7,13]]},"assertion":[{"value":"16 June 2021","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 July 2021","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}