{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T16:56:57Z","timestamp":1777654617623,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":66,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,10,17]],"date-time":"2021-10-17T00:00:00Z","timestamp":1634428800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Zhejiang Provincial Natural Science Foundation of China","award":["LR19F020004"],"award-info":[{"award-number":["LR19F020004"]}]},{"name":"Ministry of Education, National Natural Science Foundation of China","award":["U20A20222"],"award-info":[{"award-number":["U20A20222"]}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2020AAA0107400"],"award-info":[{"award-number":["2020AAA0107400"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,10,17]]},"DOI":"10.1145\/3474085.3475265","type":"proceedings-article","created":{"date-parts":[[2021,10,18]],"date-time":"2021-10-18T11:31:01Z","timestamp":1634556661000},"page":"880-889","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":25,"title":["When Video Classification Meets Incremental Classes"],"prefix":"10.1145","author":[{"given":"Hanbin","family":"Zhao","sequence":"first","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xin","family":"Qin","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shihao","family":"Su","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yongjian","family":"Fu","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zibo","family":"Lin","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xi","family":"Li","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,10,17]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/2393347.2396511"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-7908-2604-3_16"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01063"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01258-8_15"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01252-6_33"},{"key":"e_1_3_2_1_7_1","volume-title":"A continual learning survey: Defying forgetting in classification tasks","author":"Delange Matthias","year":"2021"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_9_1","volume-title":"Podnet: Pooled outputs distillation for small-tasks incremental learning. In Computer vision-ECCV 2020--16th European conference","author":"Douillard Arthur","year":"2020"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00028"},{"key":"e_1_3_2_1_11_1","volume-title":"SlowFast Networks for Video Recognition. In IEEE\/CVF International Conference on Computer Vision, ICCV. IEEE, 6201--6210","author":"Feichtenhofer Christoph","year":"2019"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.5555\/3157382.3157486"},{"key":"e_1_3_2_1_13_1","volume-title":"Spatiotemporal Multiplier Networks for Video Action Recognition. In IEEE Conference on Computer Vision and Pattern Recognition, CVPR. IEEE Computer Society, 7445--7454","author":"Feichtenhofer Christoph"},{"key":"e_1_3_2_1_14_1","volume-title":"Convolutional Two-Stream Network Fusion for Video Action Recognition. In IEEE Conference on Computer Vision and Pattern Recognition, CVPR. IEEE Computer Society","author":"Feichtenhofer Christoph","year":"2016"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.622"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00685"},{"key":"e_1_3_2_1_17_1","volume-title":"Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531","author":"Hinton Geoffrey","year":"2015"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00092"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58517-4_41"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2012.59"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2019.2962216"},{"key":"e_1_3_2_1_22_1","volume-title":"STM: SpatioTemporal and Motion Encoding for Action Recognition. In IEEE\/CVF International Conference on Computer Vision, ICCV. IEEE","author":"Jiang Boyuan","year":"2019"},{"key":"e_1_3_2_1_23_1","volume-title":"Deep generative dual memory network for continual learning. arXiv preprint arXiv:1710.10368","author":"Kamra Nitin","year":"2017"},{"key":"e_1_3_2_1_24_1","unstructured":"Will Kay Joao Carreira Karen Simonyan Brian Zhang Chloe Hillier Sudheendra Vijayanarasimhan Fabio Viola Tim Green Trevor Back Paul Natsev etal 2017. The kinetics human action video dataset. arXiv preprint arXiv:1705.06950 (2017).  Will Kay Joao Carreira Karen Simonyan Brian Zhang Chloe Hillier Sudheendra Vijayanarasimhan Fabio Viola Tim Green Trevor Back Paul Natsev et al. 2017. The kinetics human action video dataset. arXiv preprint arXiv:1705.06950 (2017)."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58517-4_21"},{"key":"e_1_3_2_1_26_1","volume-title":"IEEE Conference on Computer Vision and Pattern Recognition, CVPR. IEEE Computer Society, 204--212","author":"Lan Zhen-Zhong","year":"2015"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2818132"},{"key":"e_1_3_2_1_28_1","volume-title":"TEA: Temporal Excitation and Aggregation for Action Recognition. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR. IEEE, 906--915","author":"Li Yan","year":"2020"},{"key":"e_1_3_2_1_29_1","volume-title":"Learning without forgetting","author":"Li Zhizhong","year":"2017"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00718"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01226"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6836"},{"key":"e_1_3_2_1_33_1","volume-title":"Psychology of learning and motivation.","author":"McCloskey Michael"},{"key":"e_1_3_2_1_34_1","unstructured":"Adam Paszke Sam Gross Soumith Chintala Gregory Chanan Edward Yang Zachary DeVito Zeming Lin Alban Desmaison Luca Antiga and Adam Lerer. 2017. Automatic differentiation in pytorch. (2017).  Adam Paszke Sam Gross Soumith Chintala Gregory Chanan Edward Yang Zachary DeVito Zeming Lin Alban Desmaison Luca Antiga and Adam Lerer. 2017. Automatic differentiation in pytorch. (2017)."},{"key":"e_1_3_2_1_35_1","volume-title":"Action Recognition with Stacked Fisher Vectors. In European Conference of Computer Vision, ECCV (Lecture Notes in Computer Science","volume":"595","author":"Peng Xiaojiang","year":"2014"},{"key":"e_1_3_2_1_36_1","volume-title":"PcmNet: Position-Sensitive Context Modeling Network for Temporal Action Localization. arXiv preprint arXiv:2103.05270","author":"Qin Xin","year":"2021"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.590"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.587"},{"key":"e_1_3_2_1_39_1","volume-title":"Temporal Interlacing Network. In AAAI Conference on Artificial Intelligence. AAAI Press, 11966--11973","author":"Shao Hao","year":"2020"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.5555\/3294996.3295059"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.5555\/2968826.2968890"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58529-7_16"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00565"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00675"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2011.5995407"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2013.441"},{"key":"e_1_3_2_1_48_1","volume-title":"Appearance-and-Relation Networks for Video Classification. In IEEE Conference on Computer Vision and Pattern Recognition, CVPR. IEEE Computer Society, 1430--1439","author":"Wang Limin","year":"2018"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7299059"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"e_1_3_2_1_51_1","volume-title":"Non-Local Neural Networks. In IEEE Conference on Computer Vision and Pattern Recognition, CVPR. IEEE Computer Society, 7794--7803","author":"Wang Xiaolong","year":"2018"},{"key":"e_1_3_2_1_52_1","volume-title":"Spatiotemporal Pyramid Network for Video Action Recognition. In IEEE Conference on Computer Vision and Pattern Recognition, CVPR. IEEE Computer Society","author":"Wang Yunbo"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1145\/1553374.1553517"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2020.3001693"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00046"},{"key":"e_1_3_2_1_56_1","volume-title":"Context-aware deep spatiotemporal network for hand pose estimation from depth images","author":"Wu Yiming","year":"2018"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2964328"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01267-0_19"},{"key":"e_1_3_2_1_59_1","volume-title":"Rethinking Spatiotemporal Feature Learning: Speed-Accuracy Trade-offs in Video Classification. In European Conference of Computer Vision, ECCV (Lecture Notes in Computer Science","volume":"335","author":"Xie Saining","year":"2018"},{"key":"e_1_3_2_1_60_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 6982--6991","author":"Yu Lu"},{"key":"e_1_3_2_1_61_1","volume-title":"MgSvF: Multi-Grained Slow vs. Fast Framework for Few-Shot Class-Incremental Learning. arXiv preprint arXiv:2006.15524","author":"Zhao Hanbin","year":"2020"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2021.3072041"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.5555\/3327144.3327148"},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01246-5_49"},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00054"},{"key":"e_1_3_2_1_66_1","volume-title":"ECO: Efficient Convolutional Network for Online Video Understanding. In European Conference of Computer Vision, ECCV (Lecture Notes in Computer Science","volume":"730","author":"Zolfaghari Mohammadreza","year":"2018"}],"event":{"name":"MM '21: ACM Multimedia Conference","location":"Virtual Event China","acronym":"MM '21","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 29th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3475265","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3474085.3475265","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:48:17Z","timestamp":1750193297000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3475265"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,10,17]]},"references-count":66,"alternative-id":["10.1145\/3474085.3475265","10.1145\/3474085"],"URL":"https:\/\/doi.org\/10.1145\/3474085.3475265","relation":{},"subject":[],"published":{"date-parts":[[2021,10,17]]},"assertion":[{"value":"2021-10-17","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}