{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,24]],"date-time":"2025-03-24T08:31:35Z","timestamp":1742805095048},"reference-count":17,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"1","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2021,1,1]]},"DOI":"10.1587\/transinf.2020edl0002","type":"journal-article","created":{"date-parts":[[2020,12,31]],"date-time":"2020-12-31T22:26:22Z","timestamp":1609453582000},"page":"220-224","source":"Crossref","is-referenced-by-count":4,"title":["Spatio-Temporal Self-Attention Weighted VLAD Neural Network for Action Recognition"],"prefix":"10.1587","volume":"E104.D","author":[{"given":"Shilei","family":"CHENG","sequence":"first","affiliation":[{"name":"School of Information and Communication, University of Electronic Science and Technology of China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mei","family":"XIE","sequence":"additional","affiliation":[{"name":"School of Information and Communication, University of Electronic Science and Technology of China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zheng","family":"MA","sequence":"additional","affiliation":[{"name":"School of Information and Communication, University of Electronic Science and Technology of China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Siqi","family":"LI","sequence":"additional","affiliation":[{"name":"School of Information and Communication, University of Electronic Science and Technology of China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Song","family":"GU","sequence":"additional","affiliation":[{"name":"Department of Aircraft Maintenance Engineering, Chengdu Aeronautic Polytechnic"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Feng","family":"YANG","sequence":"additional","affiliation":[{"name":"School of Information and Communication, University of Electronic Science and Technology of China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"key":"1","doi-asserted-by":"crossref","unstructured":"[1] C. Feichtenhofer, A. Pinz, and A. Zisserman, \u201cConvolutional two-stream network fusion for video action recognition,\u201d Proc. IEEE Conf. Comput. Vis. Pattern Recognit., pp.1933-1941, 2016. 10.1109\/CVPR.2016.213","DOI":"10.1109\/CVPR.2016.213"},{"key":"2","doi-asserted-by":"crossref","unstructured":"[2] H. Wang and C. Schmid, \u201cAction recognition with improved trajectories,\u201d Proc. IEEE Int. Conf. Comput. Vis., pp.3551-3558, 2013. 10.1109\/ICCV.2013.441","DOI":"10.1109\/ICCV.2013.441"},{"key":"3","doi-asserted-by":"crossref","unstructured":"[3] A. Klaser, M. Marsza\u0142ek, and C. Schmid, \u201cA spatio-temporal descriptor based on 3d-gradients,\u201d 2008. 10.5244\/C.22.99","DOI":"10.5244\/C.22.99"},{"key":"4","doi-asserted-by":"crossref","unstructured":"[4] H. J\u00e9gou, M. Douze, C. Schmid, and P. P\u00e9rez, \u201cAggregating local descriptors into a compact image representation,\u201d CVPR 2010-23rd, IEEE Computer Society Conf. Comput. Vis. Pattern Recognit., pp.3304-3311, 2010. 10.1109\/CVPR.2010.5540039","DOI":"10.1109\/CVPR.2010.5540039"},{"key":"5","doi-asserted-by":"crossref","unstructured":"[5] L. Wang, Y. Xiong, Z. Wang, Y. Qiao, D. Lin, X. Tang, and L. Van Gool, \u201cTemporal segment networks: Towards good practices for deep action recognition,\u201d European Conference on Comput. Vis., pp.20-36, Springer, 2016. 10.1007\/978-3-319-46484-8_2","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"6","doi-asserted-by":"crossref","unstructured":"[6] D. Tran, L. Bourdev, R. Fergus, L. Torresani, and M. Paluri, \u201cLearning spatiotemporal features with 3d convolutional networks,\u201d Proc. IEEE Int. Conf. Comput. Vis., pp.4489-4497, 2015. 10.1109\/ICCV.2015.510","DOI":"10.1109\/ICCV.2015.510"},{"key":"7","unstructured":"[7] W. Kay, J. Carreira, K. Simonyan, B. Zhang, C. Hillier, S. Vijayanarasimhan, F. Viola, T. Green, T. Back, P. Natsev, M. Suleyman, and A. Zisserman, \u201cThe kinetics human action video dataset,\u201d arXiv preprint arXiv:1705.06950, 2017."},{"key":"8","doi-asserted-by":"crossref","unstructured":"[8] A. Karpathy, G. Toderici, S. Shetty, T. Leung, R. Sukthankar, and L. Fei-Fei, \u201cLarge-scale video classification with convolutional neural networks,\u201d IEEE Conf. Comput. Vis. Pattern Recognit., 2014. 10.1109\/CVPR.2014.223","DOI":"10.1109\/CVPR.2014.223"},{"key":"9","doi-asserted-by":"crossref","unstructured":"[9] R. Girdhar, D. Ramanan, A. Gupta, J. Sivic, and B. Russell, \u201cActionvlad: Learning spatio-temporal aggregation for action classification,\u201d Proc. IEEE Conf. Comput. Vis. Pattern Recognit., pp.971-980, 2017. 10.1109\/CVPR.2017.337","DOI":"10.1109\/CVPR.2017.337"},{"key":"10","doi-asserted-by":"publisher","unstructured":"[10] Z. Li, K. Gavrilyuk, E. Gavves, M. Jain, and C.G. Snoek, \u201cVideolstm convolves, attends and flows for action recognition,\u201d Computer Vision and Image Understanding, vol.166, pp.41-50, Jan. 2018. 10.1016\/j.cviu.2017.10.011","DOI":"10.1016\/j.cviu.2017.10.011"},{"key":"11","unstructured":"[11] J.Y.-H. Ng, M. Hausknecht, S. Vijayanarasimhan, O. Vinyals, R. Monga, and G. Toderici, \u201cBeyond short snippets: Deep networks for video classification,\u201d Proc. IEEE Conf. Comput. Vis. Pattern Recognit., pp.4694-4702, 2015. 10.1109\/CVPR.2015.7299101"},{"key":"12","unstructured":"[12] A. Vaswani, N. Shazeer, N. Parmar, J. Uszkoreit, L. Jones, A.N. Gomez, \u0141. Kaiser, and I. Polosukhin, \u201cAttention is all you need,\u201d Advances in Neural Information Processing Systems, pp.5998-6008, 2017."},{"key":"13","doi-asserted-by":"crossref","unstructured":"[13] X. Wang, R. Girshick, A. Gupta, and K. He, \u201cNon-local neural networks,\u201d Proc. IEEE Conf. Comput. Vis. Pattern Recognit., pp.7794-7803, 2018. 10.1109\/CVPR.2018.00813","DOI":"10.1109\/CVPR.2018.00813"},{"key":"14","unstructured":"[14] K. Soomro, A.R. Zamir, and M. Shah, \u201cUcf101: A dataset of 101 human actions classes from videos in the wild,\u201d arXiv preprint arXiv:1212.0402, 2012."},{"key":"15","doi-asserted-by":"crossref","unstructured":"[15] H. Kuehne, H. Jhuang, E. Garrote, T. Poggio, and T. Serre, \u201cHMDB: a large video database for human motion recognition,\u201d 2011 IEEE Int. Conf. Comput. Vis., pp.2556-2563, 2011. 10.1109\/ICCV.2011.6126543","DOI":"10.1109\/ICCV.2011.6126543"},{"key":"16","doi-asserted-by":"crossref","unstructured":"[16] R. Arandjelovic, P. Gronat, A. Torii, T. Pajdla, and J. Sivic, \u201cNetVLAD: Cnn architecture for weakly supervised place recognition,\u201d Proc. IEEE Conf. Comput. Vis. Pattern Recognit., pp.5297-5307, 2016. 10.1109\/CVPR.2016.572","DOI":"10.1109\/CVPR.2016.572"},{"key":"17","unstructured":"[17] G. Csurka, C.R. Dance, L. Fan, J. Willamowski, and C. Bray, \u201cVisual categorization with bags of keypoints,\u201d Workshop on Statistical Learning in Computer Vision, ECCV, pp.1-2, Prague, 2004."}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E104.D\/1\/E104.D_2020EDL0002\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,1,2]],"date-time":"2021-01-02T03:25:23Z","timestamp":1609557923000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E104.D\/1\/E104.D_2020EDL0002\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,1,1]]},"references-count":17,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2021]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2020edl0002","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"value":"0916-8532","type":"print"},{"value":"1745-1361","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,1,1]]}}}