{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:20:35Z","timestamp":1750220435901,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":76,"publisher":"ACM","license":[{"start":{"date-parts":[[2020,10,12]],"date-time":"2020-10-12T00:00:00Z","timestamp":1602460800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Natural Science Foundation of China","award":["61771440, 41776113"],"award-info":[{"award-number":["61771440, 41776113"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2020,10,12]]},"DOI":"10.1145\/3394171.3413646","type":"proceedings-article","created":{"date-parts":[[2020,10,12]],"date-time":"2020-10-12T13:10:18Z","timestamp":1602508218000},"page":"2057-2066","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Multi-Group Multi-Attention"],"prefix":"10.1145","author":[{"given":"Zhensheng","family":"Shi","sequence":"first","affiliation":[{"name":"Ocean University of China, Qingdao, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Liangjie","family":"Cao","sequence":"additional","affiliation":[{"name":"Ocean University of China, Qingdao, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Cheng","family":"Guan","sequence":"additional","affiliation":[{"name":"Ocean University of China, Qingdao, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ju","family":"Liang","sequence":"additional","affiliation":[{"name":"Ocean University of China, Qingdao, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qianqian","family":"Li","sequence":"additional","affiliation":[{"name":"Ocean University of China, Qingdao, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhaorui","family":"Gu","sequence":"additional","affiliation":[{"name":"Ocean University of China, Qingdao, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Haiyong","family":"Zheng","sequence":"additional","affiliation":[{"name":"Ocean University of China, Qingdao, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bing","family":"Zheng","sequence":"additional","affiliation":[{"name":"Ocean University of China, Qingdao, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2020,10,12]]},"reference":[{"doi-asserted-by":"crossref","unstructured":"Sagie Benaim Ariel Ephrat Oran Lang Inbar Mosseri William T Freeman Michael Rubinstein Michal Irani and Tali Dekel. 2020. SpeedNet: Learning the Speediness in Videos. In CVPR. 9922--9931.  Sagie Benaim Ariel Ephrat Oran Lang Inbar Mosseri William T Freeman Michael Rubinstein Michal Irani and Tali Dekel. 2020. SpeedNet: Learning the Speediness in Videos. In CVPR. 9922--9931.","key":"e_1_3_2_2_1_1","DOI":"10.1109\/CVPR42600.2020.00994"},{"doi-asserted-by":"crossref","unstructured":"Joao Carreira and Andrew Zisserman. 2017. Quo vadis action recognition? a new model and the Kinetics dataset. In CVPR. IEEE 6299--6308.  Joao Carreira and Andrew Zisserman. 2017. Quo vadis action recognition? a new model and the Kinetics dataset. In CVPR. IEEE 6299--6308.","key":"e_1_3_2_2_2_1","DOI":"10.1109\/CVPR.2017.502"},{"doi-asserted-by":"publisher","key":"e_1_3_2_2_3_1","DOI":"10.1145\/3123266.3123349"},{"volume-title":"MARS: Motion-Augmented RGB Stream for Action Recognition","year":"2019","author":"Crasto Nieves","key":"e_1_3_2_2_4_1"},{"doi-asserted-by":"crossref","unstructured":"Ali Diba Mohsen Fayyaz Vivek Sharma M Mahdi Arzani Rahman Yousefzadeh Juergen Gall and Luc Van Gool. 2018. Spatio-temporal channel correlation networks for action classification. In ECCV. Springer 284--299.  Ali Diba Mohsen Fayyaz Vivek Sharma M Mahdi Arzani Rahman Yousefzadeh Juergen Gall and Luc Van Gool. 2018. Spatio-temporal channel correlation networks for action classification. In ECCV. Springer 284--299.","key":"e_1_3_2_2_5_1","DOI":"10.1007\/978-3-030-01225-0_18"},{"volume-title":"NIPS. Curran Associates","author":"Feichtenhofer Christoph","key":"e_1_3_2_2_6_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_2_7_1","DOI":"10.1109\/CVPR.2016.213"},{"unstructured":"Chuang Gan Boqing Gong Kun Liu Hao Su and Leonidas J Guibas. 2018. Geometry guided convolutional neural networks for self-supervised video representation learning. In CVPR. 5589--5597.  Chuang Gan Boqing Gong Kun Liu Hao Su and Leonidas J Guibas. 2018. Geometry guided convolutional neural networks for self-supervised video representation learning. In CVPR. 5589--5597.","key":"e_1_3_2_2_8_1"},{"doi-asserted-by":"crossref","unstructured":"Chuang Gan Chen Sun Lixin Duan and Boqing Gong. 2016a. Webly-supervised video recognition by mutually voting for relevant web images and web video frames. In ECCV. Springer 849--866.  Chuang Gan Chen Sun Lixin Duan and Boqing Gong. 2016a. Webly-supervised video recognition by mutually voting for relevant web images and web video frames. In ECCV. Springer 849--866.","key":"e_1_3_2_2_9_1","DOI":"10.1007\/978-3-319-46487-9_52"},{"unstructured":"Chuang Gan Naiyan Wang Yi Yang Dit-Yan Yeung and Alex G. Hauptmann. 2015. DevNet: A Deep Event Network for Multimedia Event Detection and Evidence Recounting. In CVPR. IEEE 2568--2577.  Chuang Gan Naiyan Wang Yi Yang Dit-Yan Yeung and Alex G. Hauptmann. 2015. DevNet: A Deep Event Network for Multimedia Event Detection and Evidence Recounting. In CVPR. IEEE 2568--2577.","key":"e_1_3_2_2_10_1"},{"unstructured":"Chuang Gan Ting Yao Kuiyuan Yang Yi Yang and Tao Mei. 2016b. You lead we exceed: Labor-free video concept learning by jointly exploiting web videos and images. In CVPR. IEEE 923--932.  Chuang Gan Ting Yao Kuiyuan Yang Yi Yang and Tao Mei. 2016b. You lead we exceed: Labor-free video concept learning by jointly exploiting web videos and images. In CVPR. IEEE 923--932.","key":"e_1_3_2_2_11_1"},{"doi-asserted-by":"crossref","unstructured":"Deepti Ghadiyaram Du Tran and Dhruv Mahajan. 2019. Large-scale weakly-supervised pre-training for video action recognition. In CVPR. IEEE 12046--12055.  Deepti Ghadiyaram Du Tran and Dhruv Mahajan. 2019. Large-scale weakly-supervised pre-training for video action recognition. In CVPR. IEEE 12046--12055.","key":"e_1_3_2_2_12_1","DOI":"10.1109\/CVPR.2019.01232"},{"volume-title":"NIPS. Curran Associates","author":"Girdhar Rohit","key":"e_1_3_2_2_13_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_2_14_1","DOI":"10.1016\/0166-2236(92)90344-8"},{"unstructured":"Priya Goyal Piotr Doll\u00e1r Ross Girshick Pieter Noordhuis Lukasz Wesolowski Aapo Kyrola Andrew Tulloch Yangqing Jia and Kaiming He. 2017a. Accurate large minibatch sgd: Training imagenet in 1 hour. arXiv preprint arXiv:1706.02677 (2017).  Priya Goyal Piotr Doll\u00e1r Ross Girshick Pieter Noordhuis Lukasz Wesolowski Aapo Kyrola Andrew Tulloch Yangqing Jia and Kaiming He. 2017a. Accurate large minibatch sgd: Training imagenet in 1 hour. arXiv preprint arXiv:1706.02677 (2017).","key":"e_1_3_2_2_15_1"},{"doi-asserted-by":"crossref","unstructured":"Raghav Goyal Samira Ebrahimi Kahou Vincent Michalski Joanna Materzynska Susanne Westphal Heuna Kim Valentin Haenel Ingo Fruend Peter Yianilos Moritz Mueller-Freitag et almbox. 2017b. The \"something something\" video database for learning and evaluating visual common sense.. In ICCV. IEEE 5842--5850.  Raghav Goyal Samira Ebrahimi Kahou Vincent Michalski Joanna Materzynska Susanne Westphal Heuna Kim Valentin Haenel Ingo Fruend Peter Yianilos Moritz Mueller-Freitag et almbox. 2017b. The \"something something\" video database for learning and evaluating visual common sense.. In ICCV. IEEE 5842--5850.","key":"e_1_3_2_2_16_1","DOI":"10.1109\/ICCV.2017.622"},{"unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep residual learning for image recognition. In CVPR. IEEE 770--778.  Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep residual learning for image recognition. In CVPR. IEEE 770--778.","key":"e_1_3_2_2_17_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_2_18_1","DOI":"10.1016\/j.imavis.2017.01.010"},{"volume-title":"Proceedings of the 25th ACM international conference on Multimedia. ACM, 1087--1095","year":"2017","author":"Ruiz Alejandro Hernandez","key":"e_1_3_2_2_19_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_2_20_1","DOI":"10.1162\/neco.1997.9.8.1735"},{"unstructured":"Sergey Ioffe and Christian Szegedy. 2015. Batch normalization: Accelerating deep network training by reducing internal covariate shift. arXiv preprint arXiv:1502.03167 (2015).  Sergey Ioffe and Christian Szegedy. 2015. Batch normalization: Accelerating deep network training by reducing internal covariate shift. arXiv preprint arXiv:1502.03167 (2015).","key":"e_1_3_2_2_21_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_2_22_1","DOI":"10.1109\/TPAMI.2012.59"},{"volume-title":"S\u2122: SpatioTemporal and Motion Encoding for Action Recognition","year":"2019","author":"Jiang Boyuan","key":"e_1_3_2_2_23_1"},{"doi-asserted-by":"crossref","unstructured":"Andrej Karpathy George Toderici Sanketh Shetty Thomas Leung Rahul Sukthankar and Li Fei-Fei. 2014. Large-scale video classification with convolutional neural networks. In CVPR. IEEE 1725--1732.  Andrej Karpathy George Toderici Sanketh Shetty Thomas Leung Rahul Sukthankar and Li Fei-Fei. 2014. Large-scale video classification with convolutional neural networks. In CVPR. IEEE 1725--1732.","key":"e_1_3_2_2_24_1","DOI":"10.1109\/CVPR.2014.223"},{"unstructured":"Will Kay Joao Carreira Karen Simonyan Brian Zhang Chloe Hillier Sudheendra Vijayanarasimhan Fabio Viola Tim Green Trevor Back and Paul Natsev. 2017. The Kinetics Human Action Video Dataset. arXiv preprint arXiv:1409.1556 (2017).  Will Kay Joao Carreira Karen Simonyan Brian Zhang Chloe Hillier Sudheendra Vijayanarasimhan Fabio Viola Tim Green Trevor Back and Paul Natsev. 2017. The Kinetics Human Action Video Dataset. arXiv preprint arXiv:1409.1556 (2017).","key":"e_1_3_2_2_25_1"},{"unstructured":"Alex Krizhevsky Ilya Sutskever and Geoffrey E Hinton. 2012. ImageNet classification with deep convolutional neural networks. In NIPS. 1097--1105.  Alex Krizhevsky Ilya Sutskever and Geoffrey E Hinton. 2012. ImageNet classification with deep convolutional neural networks. In NIPS. 1097--1105.","key":"e_1_3_2_2_26_1"},{"volume-title":"HMDB: A large video database for human motion recognition","year":"2011","author":"Kuehne Hildegard","key":"e_1_3_2_2_27_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_2_28_1","DOI":"10.1109\/5.726791"},{"doi-asserted-by":"publisher","key":"e_1_3_2_2_29_1","DOI":"10.1145\/3123266.3123432"},{"unstructured":"Kunpeng Li Ziyan Wu Kuan-Chuan Peng Jan Ernst and Yun Fu. 2018. Tell me where to look: Guided attention inference network. In CVPR. IEEE 9215--9223.  Kunpeng Li Ziyan Wu Kuan-Chuan Peng Jan Ernst and Yun Fu. 2018. Tell me where to look: Guided attention inference network. In CVPR. IEEE 9215--9223.","key":"e_1_3_2_2_30_1"},{"volume-title":"TSM: Temporal shift module for efficient video understanding","year":"2019","author":"Lin Ji","key":"e_1_3_2_2_31_1"},{"unstructured":"Xingyu Liu Joon-Young Lee and Hailin Jin. 2019. Learning Video Representations from Correspondence Proposals. In CVPR. IEEE 4273--4281.  Xingyu Liu Joon-Young Lee and Hailin Jin. 2019. Learning Video Representations from Correspondence Proposals. In CVPR. IEEE 4273--4281.","key":"e_1_3_2_2_32_1"},{"doi-asserted-by":"crossref","unstructured":"Xiang Long Chuang Gan Gerard de Melo Jiajun Wu Xiao Liu and Shilei Wen. 2018. Attention clusters: Purely attention based local feature integration for video classification. In CVPR. IEEE 7834--7843.  Xiang Long Chuang Gan Gerard de Melo Jiajun Wu Xiao Liu and Shilei Wen. 2018. Attention clusters: Purely attention based local feature integration for video classification. In CVPR. IEEE 7834--7843.","key":"e_1_3_2_2_33_1","DOI":"10.1109\/CVPR.2018.00817"},{"unstructured":"Chenxu Luo and Alan L Yuille. 2019. Grouped Spatial-Temporal Aggregation for Efficient Action Recognition. In ICCV. IEEE 5512--5521.  Chenxu Luo and Alan L Yuille. 2019. Grouped Spatial-Temporal Aggregation for Efficient Action Recognition. In ICCV. IEEE 5512--5521.","key":"e_1_3_2_2_34_1"},{"unstructured":"Chih-Yao Ma Asim Kadav Iain Melvin Zsolt Kira Ghassan AlRegib and Hans Peter Graf. 2018a. Attend and Interact: Higher-Order Object Interactions for Video Understanding. In CVPR. IEEE 6790--6800.  Chih-Yao Ma Asim Kadav Iain Melvin Zsolt Kira Ghassan AlRegib and Hans Peter Graf. 2018a. Attend and Interact: Higher-Order Object Interactions for Video Understanding. In CVPR. IEEE 6790--6800.","key":"e_1_3_2_2_35_1"},{"unstructured":"Ningning Ma Xiangyu Zhang Hai-Tao Zheng and Jian Sun. 2018b. Shufflenet v2: Practical guidelines for efficient cnn architecture design. In ECCV. Springer 116--131.  Ningning Ma Xiangyu Zhang Hai-Tao Zheng and Jian Sun. 2018b. Shufflenet v2: Practical guidelines for efficient cnn architecture design. In ECCV. Springer 116--131.","key":"e_1_3_2_2_36_1"},{"doi-asserted-by":"crossref","unstructured":"Brais Martinez Davide Modolo Yuanjun Xiong and Joseph Tighe. 2019. Action Recognition With Spatial-Temporal Discriminative Filter Banks. In ICCV. IEEE 5482--5491.  Brais Martinez Davide Modolo Yuanjun Xiong and Joseph Tighe. 2019. Action Recognition With Spatial-Temporal Discriminative Filter Banks. In ICCV. IEEE 5482--5491.","key":"e_1_3_2_2_37_1","DOI":"10.1109\/ICCV.2019.00558"},{"unstructured":"Lili Meng Bo Zhao Bo Chang Gao Huang Frederick Tung and Leonid Sigal. 2018. Where and When to Look? Spatio-temporal Attention for Action Recognition in Videos. arXiv preprint arXiv:1810.04511 (2018).  Lili Meng Bo Zhao Bo Chang Gao Huang Frederick Tung and Leonid Sigal. 2018. Where and When to Look? Spatio-temporal Attention for Action Recognition in Videos. arXiv preprint arXiv:1810.04511 (2018).","key":"e_1_3_2_2_38_1"},{"unstructured":"Volodymyr Mnih Nicolas Heess Alex Graves et almbox. 2014. Recurrent models of visual attention. In NIPS. Curran Associates Inc. 2204--2212.  Volodymyr Mnih Nicolas Heess Alex Graves et almbox. 2014. Recurrent models of visual attention. In NIPS. Curran Associates Inc. 2204--2212.","key":"e_1_3_2_2_39_1"},{"unstructured":"Zhaofan Qiu Ting Yao and Tao Mei. 2017. Learning spatio-temporal representation with Pseudo-3D residual networks. In ICCV. IEEE 5533--5541.  Zhaofan Qiu Ting Yao and Tao Mei. 2017. Learning spatio-temporal representation with Pseudo-3D residual networks. In ICCV. IEEE 5533--5541.","key":"e_1_3_2_2_40_1"},{"unstructured":"Zhaofan Qiu Ting Yao Chong-Wah Ngo Xinmei Tian and Tao Mei. 2019. Learning spatio-temporal representation with local and global diffusion. In CVPR. IEEE 12056--12065.  Zhaofan Qiu Ting Yao Chong-Wah Ngo Xinmei Tian and Tao Mei. 2019. Learning spatio-temporal representation with local and global diffusion. In CVPR. IEEE 12056--12065.","key":"e_1_3_2_2_41_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_2_42_1","DOI":"10.1080\/135062800394667"},{"doi-asserted-by":"publisher","key":"e_1_3_2_2_43_1","DOI":"10.1016\/j.tics.2005.11.008"},{"unstructured":"Shikhar Sharma Ryan Kiros and Ruslan Salakhutdinov. 2015. Action recognition using visual attention. arXiv preprint arXiv:1511.04119 (2015).  Shikhar Sharma Ryan Kiros and Ruslan Salakhutdinov. 2015. Action recognition using visual attention. arXiv preprint arXiv:1511.04119 (2015).","key":"e_1_3_2_2_44_1"},{"volume-title":"NIPS. Curran Associates","author":"Simonyan Karen","key":"e_1_3_2_2_45_1"},{"unstructured":"Karen Simonyan and Andrew Zisserman. 2014b. Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014).  Karen Simonyan and Andrew Zisserman. 2014b. Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014).","key":"e_1_3_2_2_46_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_2_47_1","DOI":"10.1145\/3240508.3240713"},{"unstructured":"Khurram Soomro Amir Roshan Zamir and Mubarak Shah. 2012. UCF101: A dataset of 101 human actions classes from videos in the wild. arXiv preprint arXiv:1212.0402 (2012).  Khurram Soomro Amir Roshan Zamir and Mubarak Shah. 2012. UCF101: A dataset of 101 human actions classes from videos in the wild. arXiv preprint arXiv:1212.0402 (2012).","key":"e_1_3_2_2_48_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_2_49_1","DOI":"10.1037\/0022-0663.75.4.471"},{"doi-asserted-by":"crossref","unstructured":"Ming Sun Yuchen Yuan Feng Zhou and Errui Ding. 2018. Multi-Attention Multi-Class Constraint for Fine-grained Image Recognition. In ECCV. Springer 805--821.  Ming Sun Yuchen Yuan Feng Zhou and Errui Ding. 2018. Multi-Attention Multi-Class Constraint for Fine-grained Image Recognition. In ECCV. Springer 805--821.","key":"e_1_3_2_2_50_1","DOI":"10.1007\/978-3-030-01270-0_49"},{"doi-asserted-by":"crossref","unstructured":"Christian Szegedy Sergey Ioffe Vincent Vanhoucke and Alexander A Alemi. 2017. Inception-v4 Inception-ResNet and the impact of residual connections on learning. In AAAI. AAAIPress 4278--4284.  Christian Szegedy Sergey Ioffe Vincent Vanhoucke and Alexander A Alemi. 2017. Inception-v4 Inception-ResNet and the impact of residual connections on learning. In AAAI. AAAIPress 4278--4284.","key":"e_1_3_2_2_51_1","DOI":"10.1609\/aaai.v31i1.11231"},{"doi-asserted-by":"crossref","unstructured":"Christian Szegedy Wei Liu Yangqing Jia Pierre Sermanet Scott Reed Dragomir Anguelov Dumitru Erhan Vincent Vanhoucke and Andrew Rabinovich. 2015. Going deeper with convolutions. In CVPR. IEEE 1--9.  Christian Szegedy Wei Liu Yangqing Jia Pierre Sermanet Scott Reed Dragomir Anguelov Dumitru Erhan Vincent Vanhoucke and Andrew Rabinovich. 2015. Going deeper with convolutions. In CVPR. IEEE 1--9.","key":"e_1_3_2_2_52_1","DOI":"10.1109\/CVPR.2015.7298594"},{"doi-asserted-by":"crossref","unstructured":"Christian Szegedy Vincent Vanhoucke Sergey Ioffe Jon Shlens and Zbigniew Wojna. 2016. Rethinking the inception architecture for computer vision. In CVPR. IEEE 2818--2826.  Christian Szegedy Vincent Vanhoucke Sergey Ioffe Jon Shlens and Zbigniew Wojna. 2016. Rethinking the inception architecture for computer vision. In CVPR. IEEE 2818--2826.","key":"e_1_3_2_2_53_1","DOI":"10.1109\/CVPR.2016.308"},{"doi-asserted-by":"crossref","unstructured":"Du Tran Lubomir Bourdev Rob Fergus Lorenzo Torresani and Manohar Paluri. 2015. Learning spatiotemporal features with 3D convolutional networks. In ICCV. IEEE 4489--4497.  Du Tran Lubomir Bourdev Rob Fergus Lorenzo Torresani and Manohar Paluri. 2015. Learning spatiotemporal features with 3D convolutional networks. In ICCV. IEEE 4489--4497.","key":"e_1_3_2_2_54_1","DOI":"10.1109\/ICCV.2015.510"},{"doi-asserted-by":"crossref","unstructured":"Du Tran Heng Wang Lorenzo Torresani and Matt Feiszli. 2019. Video classification with channel-separated convolutional networks. In ICCV. IEEE 5552--5561.  Du Tran Heng Wang Lorenzo Torresani and Matt Feiszli. 2019. Video classification with channel-separated convolutional networks. In ICCV. IEEE 5552--5561.","key":"e_1_3_2_2_55_1","DOI":"10.1109\/ICCV.2019.00565"},{"doi-asserted-by":"crossref","unstructured":"Du Tran Heng Wang Lorenzo Torresani Jamie Ray Yann LeCun and Manohar Paluri. 2018. A Closer Look at Spatiotemporal Convolutions for Action Recognition. In CVPR. IEEE 6450--6459.  Du Tran Heng Wang Lorenzo Torresani Jamie Ray Yann LeCun and Manohar Paluri. 2018. A Closer Look at Spatiotemporal Convolutions for Action Recognition. In CVPR. IEEE 6450--6459.","key":"e_1_3_2_2_56_1","DOI":"10.1109\/CVPR.2018.00675"},{"volume-title":"NIPS. Curran Associates","author":"Vaswani Ashish","key":"e_1_3_2_2_57_1"},{"doi-asserted-by":"crossref","unstructured":"Fei Wang Mengqing Jiang Chen Qian Shuo Yang Cheng Li Honggang Zhang Xiaogang Wang and Xiaoou Tang. 2017. Residual attention network for image classification. In CVPR. IEEE 3156--3164.  Fei Wang Mengqing Jiang Chen Qian Shuo Yang Cheng Li Honggang Zhang Xiaogang Wang and Xiaoou Tang. 2017. Residual attention network for image classification. In CVPR. IEEE 3156--3164.","key":"e_1_3_2_2_58_1","DOI":"10.1109\/CVPR.2017.683"},{"doi-asserted-by":"crossref","unstructured":"Heng Wang Alexander Kl\u00e4ser Cordelia Schmid and Cheng-Lin Liu. 2011. Action recognition by dense trajectories. In CVPR. IEEE 3169--3676.  Heng Wang Alexander Kl\u00e4ser Cordelia Schmid and Cheng-Lin Liu. 2011. Action recognition by dense trajectories. In CVPR. IEEE 3169--3676.","key":"e_1_3_2_2_59_1","DOI":"10.1109\/CVPR.2011.5995407"},{"doi-asserted-by":"crossref","unstructured":"Heng Wang and Cordelia Schmid. 2013. Action recognition with improved trajectories. In ICCV. IEEE 3551--3558.  Heng Wang and Cordelia Schmid. 2013. Action recognition with improved trajectories. In ICCV. IEEE 3551--3558.","key":"e_1_3_2_2_60_1","DOI":"10.1109\/ICCV.2013.441"},{"doi-asserted-by":"crossref","unstructured":"Limin Wang Wei Li Wen Li and Luc Van Gool. 2018b. Appearance-and-relation networks for video classification. In CVPR. IEEE 1430--1439.  Limin Wang Wei Li Wen Li and Luc Van Gool. 2018b. Appearance-and-relation networks for video classification. In CVPR. IEEE 1430--1439.","key":"e_1_3_2_2_61_1","DOI":"10.1109\/CVPR.2018.00155"},{"doi-asserted-by":"crossref","unstructured":"Limin Wang Yuanjun Xiong Zhe Wang Yu Qiao Dahua Lin Xiaoou Tang and Luc Van Gool. 2016b. Temporal segment networks: Towards good practices for deep action recognition. In ECCV. Springer 20--36.  Limin Wang Yuanjun Xiong Zhe Wang Yu Qiao Dahua Lin Xiaoou Tang and Luc Van Gool. 2016b. Temporal segment networks: Towards good practices for deep action recognition. In ECCV. Springer 20--36.","key":"e_1_3_2_2_62_1","DOI":"10.1007\/978-3-319-46484-8_2"},{"doi-asserted-by":"publisher","key":"e_1_3_2_2_63_1","DOI":"10.1145\/2964284.2967191"},{"doi-asserted-by":"crossref","unstructured":"Xiaolong Wang Ross Girshick Abhinav Gupta and Kaiming He. 2018a. Non-local neural networks. In CVPR. IEEE 7794--7803.  Xiaolong Wang Ross Girshick Abhinav Gupta and Kaiming He. 2018a. Non-local neural networks. In CVPR. IEEE 7794--7803.","key":"e_1_3_2_2_64_1","DOI":"10.1109\/CVPR.2018.00813"},{"doi-asserted-by":"crossref","unstructured":"Xiaolong Wang and Abhinav Gupta. 2018. Videos as space-time region graphs. In ECCV. Springer 399--417.  Xiaolong Wang and Abhinav Gupta. 2018. Videos as space-time region graphs. In ECCV. Springer 399--417.","key":"e_1_3_2_2_65_1","DOI":"10.1007\/978-3-030-01228-1_25"},{"doi-asserted-by":"crossref","unstructured":"Di Wu Nabin Sharma and Michael Blumenstein. 2017. Recent advances in video-based human action recognition using deep learning: a review. In IJCNN. IEEE 2865--2872.  Di Wu Nabin Sharma and Michael Blumenstein. 2017. Recent advances in video-based human action recognition using deep learning: a review. In IJCNN. IEEE 2865--2872.","key":"e_1_3_2_2_66_1","DOI":"10.1109\/IJCNN.2017.7966210"},{"doi-asserted-by":"crossref","unstructured":"Tete Xiao Quanfu Fan Dan Gutfreund Mathew Monfort Aude Oliva and Bolei Zhou. 2019. Reasoning About Human-Object Interactions Through Dual Attention Networks. In ICCV. IEEE 3919--3928.  Tete Xiao Quanfu Fan Dan Gutfreund Mathew Monfort Aude Oliva and Bolei Zhou. 2019. Reasoning About Human-Object Interactions Through Dual Attention Networks. In ICCV. IEEE 3919--3928.","key":"e_1_3_2_2_67_1","DOI":"10.1109\/ICCV.2019.00402"},{"unstructured":"Saining Xie Ross Girshick Piotr Doll\u00e1r Zhuowen Tu and Kaiming He. 2017. Aggregated residual transformations for deep neural networks. In CVPR. IEEE 1492--1500.  Saining Xie Ross Girshick Piotr Doll\u00e1r Zhuowen Tu and Kaiming He. 2017. Aggregated residual transformations for deep neural networks. In CVPR. IEEE 1492--1500.","key":"e_1_3_2_2_68_1"},{"unstructured":"Saining Xie Chen Sun Jonathan Huang Zhuowen Tu and Kevin Murphy. 2018. Rethinking spatiotemporal feature learning: Speed-accuracy trade-offs in video classification. In ECCV. Springer 305--321.  Saining Xie Chen Sun Jonathan Huang Zhuowen Tu and Kevin Murphy. 2018. Rethinking spatiotemporal feature learning: Speed-accuracy trade-offs in video classification. In ECCV. Springer 305--321.","key":"e_1_3_2_2_69_1"},{"doi-asserted-by":"crossref","unstructured":"Joe Yue-Hei Ng Matthew Hausknecht Sudheendra Vijayanarasimhan Oriol Vinyals Rajat Monga and George Toderici. 2015. Beyond short snippets: Deep networks for video classification. In CVPR. IEEE 4694--4702.  Joe Yue-Hei Ng Matthew Hausknecht Sudheendra Vijayanarasimhan Oriol Vinyals Rajat Monga and George Toderici. 2015. Beyond short snippets: Deep networks for video classification. In CVPR. IEEE 4694--4702.","key":"e_1_3_2_2_70_1","DOI":"10.1109\/CVPR.2015.7299101"},{"volume-title":"Shufflenet: An extremely efficient convolutional neural network for mobile devices","year":"2018","author":"Zhang Xiangyu","key":"e_1_3_2_2_71_1"},{"doi-asserted-by":"crossref","unstructured":"Yue Zhao Yuanjun Xiong and Dahua Lin. 2018a. Recognize Actions by Disentangling Components of Dynamics. In CVPR. IEEE 6566--6575.  Yue Zhao Yuanjun Xiong and Dahua Lin. 2018a. Recognize Actions by Disentangling Components of Dynamics. In CVPR. IEEE 6566--6575.","key":"e_1_3_2_2_72_1","DOI":"10.1109\/CVPR.2018.00687"},{"unstructured":"Yue Zhao Yuanjun Xiong and Dahua Lin. 2018b. Trajectory convolution for action recognition. In NeurIPS. 2208--2219.  Yue Zhao Yuanjun Xiong and Dahua Lin. 2018b. Trajectory convolution for action recognition. In NeurIPS. 2208--2219.","key":"e_1_3_2_2_73_1"},{"doi-asserted-by":"crossref","unstructured":"Heliang Zheng Jianlong Fu Tao Mei and Jiebo Luo. 2017. Learning multi-attention convolutional neural network for fine-grained image recognition. In ICCV. IEEE 5209--5217.  Heliang Zheng Jianlong Fu Tao Mei and Jiebo Luo. 2017. Learning multi-attention convolutional neural network for fine-grained image recognition. In ICCV. IEEE 5209--5217.","key":"e_1_3_2_2_74_1","DOI":"10.1109\/ICCV.2017.557"},{"doi-asserted-by":"crossref","unstructured":"Bolei Zhou Alex Andonian Aude Oliva and Antonio Torralba. 2018. Temporal relational reasoning in videos. In ECCV. Springer 803--818.  Bolei Zhou Alex Andonian Aude Oliva and Antonio Torralba. 2018. Temporal relational reasoning in videos. In ECCV. Springer 803--818.","key":"e_1_3_2_2_75_1","DOI":"10.1007\/978-3-030-01246-5_49"},{"volume-title":"ECO: Efficient convolutional network for online video understanding","year":"2018","author":"Zolfaghari Mohammadreza","key":"e_1_3_2_2_76_1"}],"event":{"sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"acronym":"MM '20","name":"MM '20: The 28th ACM International Conference on Multimedia","location":"Seattle WA USA"},"container-title":["Proceedings of the 28th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3394171.3413646","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3394171.3413646","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:47:15Z","timestamp":1750193235000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3394171.3413646"}},"subtitle":["Towards Discriminative Spatiotemporal Representation"],"short-title":[],"issued":{"date-parts":[[2020,10,12]]},"references-count":76,"alternative-id":["10.1145\/3394171.3413646","10.1145\/3394171"],"URL":"https:\/\/doi.org\/10.1145\/3394171.3413646","relation":{},"subject":[],"published":{"date-parts":[[2020,10,12]]},"assertion":[{"value":"2020-10-12","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}