{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,13]],"date-time":"2026-02-13T09:44:22Z","timestamp":1770975862193,"version":"3.50.1"},"reference-count":38,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2022,9,14]],"date-time":"2022-09-14T00:00:00Z","timestamp":1663113600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,9,14]],"date-time":"2022-09-14T00:00:00Z","timestamp":1663113600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/100006502","name":"Defense Sciences Office, DARPA","doi-asserted-by":"publisher","award":["crisp\/cbric"],"award-info":[{"award-number":["crisp\/cbric"]}],"id":[{"id":"10.13039\/100006502","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Circuits Syst Signal Process"],"published-print":{"date-parts":[[2023,2]]},"DOI":"10.1007\/s00034-022-02160-x","type":"journal-article","created":{"date-parts":[[2022,9,14]],"date-time":"2022-09-14T11:03:00Z","timestamp":1663153380000},"page":"705-723","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["STAR: Efficient SpatioTemporal Modeling for Action Recognition"],"prefix":"10.1007","volume":"42","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1821-9313","authenticated-orcid":false,"given":"Abhijeet","family":"Kumar","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Samuel","family":"Abrams","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Abhishek","family":"Kumar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Vijaykrishnan","family":"Narayanan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,9,14]]},"reference":[{"issue":"2","key":"2160_CR1","doi-asserted-by":"publisher","first-page":"1043","DOI":"10.1007\/s11042-014-2345-z","volume":"75","author":"RV Babu","year":"2016","unstructured":"R.V. Babu, M. Tom, P. Wadekar, A survey on compressed domain video analysis techniques. Multimedia Tools Appl. 75(2), 1043\u20131078 (2016). https:\/\/doi.org\/10.1007\/s11042-014-2345-z","journal-title":"Multimedia Tools Appl."},{"key":"2160_CR2","doi-asserted-by":"crossref","unstructured":"B. Battash, H. Barad, H. Tang, A. Bleiweiss, Mimic the raw domain: accelerating action recognition in the compressed domain, in 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW), pp. 32926\u20132934 (2020)","DOI":"10.1109\/CVPRW50498.2020.00350"},{"key":"2160_CR3","unstructured":"H. Cao, S. Yu, J. Feng, Compressed Video Action Recognition with Refined Motion Vector (2019)"},{"key":"2160_CR4","doi-asserted-by":"crossref","unstructured":"J. Carreira, A. Zisserman, Quo Vadis, Action Recognition? A New Model and the Kinetics Dataset (2018)","DOI":"10.1109\/CVPR.2017.502"},{"key":"2160_CR5","unstructured":"Y. Chen, Y. Kalantidis, J. Li, S. Yan, J. Feng, Two-stream convolutional networks for action recognition in videos. arxiv:1807.11195 (2018)"},{"key":"2160_CR6","unstructured":"Cisco Visual Networking Index: Forecast and Trends, 20172022 White Paper\u2014Cisco (2019)"},{"issue":"1","key":"2160_CR7","doi-asserted-by":"publisher","first-page":"20","DOI":"10.1016\/0166-2236(92)90344-8","volume":"15","author":"MA Goodale","year":"1992","unstructured":"M.A. Goodale, A.D. Milner, Separate visual pathways for perception and action. Trends Neurosci. 15(1), 20\u201325 (1992)","journal-title":"Trends Neurosci."},{"key":"2160_CR8","doi-asserted-by":"crossref","unstructured":"K. He, X. Zhang, S. Ren, J. Sun, Deep Residual Learning for Image Recognition (2015)","DOI":"10.1109\/CVPR.2016.90"},{"key":"2160_CR9","unstructured":"A.G. Howard, M. Zhu, B. Chen, D. Kalenichenko, W. Wang, T. Weyand, M. Andreetto, H. Adam, Mobilenets: efficient convolutional neural networks for mobile vision applications. CoRR arxiv:1704.04861 (2017)"},{"issue":"3s","key":"2160_CR10","doi-asserted-by":"publisher","first-page":"66","DOI":"10.1145\/3422360","volume":"16","author":"H Hu","year":"2021","unstructured":"H. Hu, W. Zhou, X. Li, N. Yan, H. Li, Mv2flow: learning motion representation for fast compressed video action recognition. ACM Trans. Multimedia Comput. Commun. Appl. 16(3s), 66 (2021). https:\/\/doi.org\/10.1145\/3422360","journal-title":"ACM Trans. Multimedia Comput. Commun. Appl."},{"key":"2160_CR11","unstructured":"S. Ioffe, C. Szegedy, Batch Normalization: Accelerating Deep Network Training by Reducing Internal Covariate Shift (2015)"},{"key":"2160_CR12","unstructured":"S. Ioffe, C. Szegedy, Batch normalization: accelerating deep network training by reducing internal covariate shift. CoRR arxiv:1502.03167 (2015)"},{"issue":"1","key":"2160_CR13","doi-asserted-by":"publisher","first-page":"221","DOI":"10.1109\/TPAMI.2012.59","volume":"35","author":"S Ji","year":"2013","unstructured":"S. Ji, W. Xu, M. Yang, K. Yu, 3d convolutional neural networks for human action recognition. IEEE Trans. Pattern Anal. Mach. Intell. 35(1), 221\u2013231 (2013). https:\/\/doi.org\/10.1109\/TPAMI.2012.59","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"2160_CR14","doi-asserted-by":"publisher","unstructured":"A. Karpathy, G. Toderici, S. Shetty, T. Leung, R. Sukthankar, L. Fei-Fei, Large-scale video classification with convolutional neural networks, in 2014 IEEE Conference on Computer Vision and Pattern Recognition, pp. 1725\u20131732 (2014). https:\/\/doi.org\/10.1109\/CVPR.2014.223","DOI":"10.1109\/CVPR.2014.223"},{"key":"2160_CR15","doi-asserted-by":"publisher","unstructured":"A. Karpathy, G. Toderici, S. Shetty, T. Leung, R. Sukthankar, L. Fei-Fei, Large-scale video classification with convolutional neural networks, in 2014 IEEE Conference on Computer Vision and Pattern Recognition, pp. 1725\u20131732 (2014). https:\/\/doi.org\/10.1109\/CVPR.2014.223","DOI":"10.1109\/CVPR.2014.223"},{"key":"2160_CR16","unstructured":"W. Kay, J. Carreira, K. Simonyan, B. Zhang, C. Hillier, S. Vijayanarasimhan, F. Viola, T. Green, T. Back, P. Natsev, M. Suleyman, A. Zisserman, The Kinetics Human Action Video Dataset (2017)"},{"key":"2160_CR17","unstructured":"D.P. Kingma, J. Ba, Adam: A Method for Stochastic Optimization (2017)"},{"issue":"6","key":"2160_CR18","doi-asserted-by":"publisher","first-page":"84","DOI":"10.1145\/3065386","volume":"60","author":"A Krizhevsky","year":"2017","unstructured":"A. Krizhevsky, I. Sutskever, G.E. Hinton, Imagenet classification with deep convolutional neural networks. Commun. ACM 60(6), 84\u201390 (2017). https:\/\/doi.org\/10.1145\/3065386","journal-title":"Commun. ACM"},{"key":"2160_CR19","doi-asserted-by":"crossref","unstructured":"H. Kuehne, H. Jhuang, E. Garrote, T. Poggio, T. Serre, HMDB: a large video database for human motion recognition, in Proceedings of the International Conference on Computer Vision (ICCV) (2011)","DOI":"10.1109\/ICCV.2011.6126543"},{"key":"2160_CR20","doi-asserted-by":"crossref","unstructured":"J. Lin, C. Gan, S. Han, Temporal shift module for efficient video understanding. CoRR arxiv:1811.08383 (2018)","DOI":"10.1109\/ICCV.2019.00718"},{"key":"2160_CR21","doi-asserted-by":"publisher","unstructured":"M.K. Mandal, Digital Video Compression Techniques, pp. 203\u2013237 (Springer, Boston, 2003). https:\/\/doi.org\/10.1007\/978-1-4615-0265-4_9","DOI":"10.1007\/978-1-4615-0265-4_9"},{"key":"2160_CR22","doi-asserted-by":"publisher","unstructured":"S. Ranjbar\u00a0Alvar, H. Choi, I.V. Bajic, Can you tell a face from a hevc bitstream? in 2018 IEEE Conference on Multimedia Information Processing and Retrieval (MIPR), pp. 257\u2013261 (2018). https:\/\/doi.org\/10.1109\/MIPR.2018.00060","DOI":"10.1109\/MIPR.2018.00060"},{"key":"2160_CR23","doi-asserted-by":"crossref","unstructured":"Z. Shou, X. Lin, Y. Kalantidis, L. Sevilla-Lara, M. Rohrbach, S.-F. Chang, Z. Yan, DMC-Net: Generating Discriminative Motion Cues for Fast Compressed Video Action Recognition (2019)","DOI":"10.1109\/CVPR.2019.00136"},{"key":"2160_CR24","unstructured":"K. Simonyan, A. Zisserman, Two-Stream Convolutional Networks for Action Recognition in Videos (2014)"},{"key":"2160_CR25","unstructured":"K. Soomro, A.R. Zamir, M. Shah, UCF101: A Dataset of 101 Human Actions Classes From Videos in The Wild (2012)"},{"issue":"146","key":"2160_CR26","first-page":"10","volume":"2006","author":"S Tomar","year":"2006","unstructured":"S. Tomar, Converting video formats with ffmpeg. Linux J. 2006(146), 10 (2006)","journal-title":"Linux J."},{"key":"2160_CR27","doi-asserted-by":"crossref","unstructured":"D. Tran, L. Bourdev, R. Fergus, L. Torresani, M. Paluri, Learning Spatiotemporal Features with 3D Convolutional Networks (2015)","DOI":"10.1109\/ICCV.2015.510"},{"key":"2160_CR28","unstructured":"H. Wang, B. Raj, On the Origin of Deep Learning (2017)"},{"key":"2160_CR29","doi-asserted-by":"publisher","unstructured":"H. Wang, C. Schmid, Action recognition with improved trajectories, in 2013 IEEE International Conference on Computer Vision, pp. 3551\u20133558 (2013). https:\/\/doi.org\/10.1109\/ICCV.2013.441","DOI":"10.1109\/ICCV.2013.441"},{"key":"2160_CR30","unstructured":"L. Wang, Y. Xiong, Z. Wang, Y. Qiao, Towards Good Practices for Very Deep Two-Stream ConvNets (2015)"},{"key":"2160_CR31","doi-asserted-by":"crossref","unstructured":"L. Wang, Y. Xiong, Z. Wang, Y. Qiao, D. Lin, X. Tang, L.V. Gool Temporal Segment Networks: Towards Good Practices for Deep Action Recognition (2016)","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"2160_CR32","unstructured":"C.-Y. Wu, M. Zaheer, H. Hu, R. Manmatha, A.J. Smola, P. Kr\u00e4henb\u00fchl, Compressed Video Action Recognition (2018)"},{"key":"2160_CR33","unstructured":"C.-Y. Wu, M. Zaheer, H. Hu, R. Manmatha, A.J. Smola, P. Kr\u00e4henb\u00fchl, Compressed Video Action Recognition (2018)"},{"key":"2160_CR34","doi-asserted-by":"crossref","unstructured":"Z. Xu, Y. Yang, A.G. Hauptmann, A discriminative CNN video representation for event detection. CoRR arxiv:1411.4006 (2014)","DOI":"10.1109\/CVPR.2015.7298789"},{"key":"2160_CR35","doi-asserted-by":"publisher","unstructured":"G. Yao, T. Lei, J. Zhong, A review of convolutional-neural-network-based action recognition. Pattern Recognit. Lett. 118, 14\u201322 (2019). https:\/\/doi.org\/10.1016\/j.patrec.2018.05.018; Cooperative and Social Robots: Understanding Human Activities and Intentions","DOI":"10.1016\/j.patrec.2018.05.018"},{"key":"2160_CR36","doi-asserted-by":"crossref","unstructured":"L. Yao, A. Torabi, K. Cho, N. Ballas, C. Pal, H. Larochelle, A. Courville, Describing Videos by Exploiting Temporal Structure (2015)","DOI":"10.1109\/ICCV.2015.512"},{"key":"2160_CR37","doi-asserted-by":"publisher","first-page":"701","DOI":"10.1007\/978-3-030-68763-2_53","volume-title":"Pattern Recognition. ICPR International Workshops and Challenges","author":"C Zhou","year":"2021","unstructured":"C. Zhou, X. Chen, P. Sun, G. Zhang, W. Zhou, Compressed video action recognition using motion vector representation, in Pattern Recognition. ICPR International Workshops and Challenges. ed. by A. Del Bimbo, R. Cucchiara, S. Sclaroff, G.M. Farinella, T. Mei, M. Bertini, H.J. Escalante, R. Vezzani (Springer, Cham, 2021), pp.701\u2013713"},{"key":"2160_CR38","unstructured":"Y. Zhu, X. Li, C. Liu, M. Zolfaghari, Y. Xiong, C. Wu, Z. Zhang, J. Tighe, R. Manmatha, M. Li, A Comprehensive Study of Deep Video Action Recognition (2020)"}],"container-title":["Circuits, Systems, and Signal Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00034-022-02160-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00034-022-02160-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00034-022-02160-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,2,3]],"date-time":"2023-02-03T03:07:44Z","timestamp":1675393664000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00034-022-02160-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,9,14]]},"references-count":38,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2023,2]]}},"alternative-id":["2160"],"URL":"https:\/\/doi.org\/10.1007\/s00034-022-02160-x","relation":{},"ISSN":["0278-081X","1531-5878"],"issn-type":[{"value":"0278-081X","type":"print"},{"value":"1531-5878","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,9,14]]},"assertion":[{"value":"31 January 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 August 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 August 2022","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 September 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}