{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,26]],"date-time":"2025-10-26T00:29:52Z","timestamp":1761438592463,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":43,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,10,10]],"date-time":"2022-10-10T00:00:00Z","timestamp":1665360000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Key Research and Development Program of China Grant","award":["No.2018AAA0100400"],"award-info":[{"award-number":["No.2018AAA0100400"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["No. U21B2013, 61971277"],"award-info":[{"award-number":["No. U21B2013, 61971277"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,10,10]]},"DOI":"10.1145\/3503161.3547938","type":"proceedings-article","created":{"date-parts":[[2022,10,10]],"date-time":"2022-10-10T15:42:46Z","timestamp":1665416566000},"page":"6230-6240","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":9,"title":["Task-adaptive Spatial-Temporal Video Sampler for Few-shot Action Recognition"],"prefix":"10.1145","author":[{"given":"Huabin","family":"Liu","sequence":"first","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weixian","family":"Lv","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"John","family":"See","sequence":"additional","affiliation":[{"name":"Heriot-Watt University Malaysia, Putrajaya, Malaysia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weiyao","family":"Lin","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2022,10,10]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Learning with differentiable pertubed optimizers. Advances in neural information processing systems","author":"Berthet Quentin","year":"2020","unstructured":"Quentin Berthet , Mathieu Blondel , Olivier Teboul , Marco Cuturi , Jean-Philippe Vert , and Francis Bach . 2020. Learning with differentiable pertubed optimizers. Advances in neural information processing systems , Vol. 33 ( 2020 ), 9508--9519. Quentin Berthet, Mathieu Blondel, Olivier Teboul, Marco Cuturi, Jean-Philippe Vert, and Francis Bach. 2020. Learning with differentiable pertubed optimizers. Advances in neural information processing systems , Vol. 33 (2020), 9508--9519."},{"key":"e_1_3_2_2_2_1","volume-title":"Tarn: Temporal attentive relation network for few-shot and zero-shot action recognition. arXiv preprint arXiv:1907.09021","author":"Bishay Mina","year":"2019","unstructured":"Mina Bishay , Georgios Zoumpourlis , and Ioannis Patras . 2019 . Tarn: Temporal attentive relation network for few-shot and zero-shot action recognition. arXiv preprint arXiv:1907.09021 (2019). Mina Bishay, Georgios Zoumpourlis, and Ioannis Patras. 2019. Tarn: Temporal attentive relation network for few-shot and zero-shot action recognition. arXiv preprint arXiv:1907.09021 (2019)."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01063"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00238"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/318242.318443"},{"key":"e_1_3_2_2_8_1","volume-title":"Towards a neural statistician. arXiv preprint arXiv:1606.02185","author":"Edwards Harrison","year":"2016","unstructured":"Harrison Edwards and Amos Storkey . 2016. Towards a neural statistician. arXiv preprint arXiv:1606.02185 ( 2016 ). Harrison Edwards and Amos Storkey. 2016. Towards a neural statistician. arXiv preprint arXiv:1606.02185 (2016)."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.476"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01535"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.622"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_2_13_1","volume-title":"Spatial transformer networks. arXiv preprint arXiv:1506.02025","author":"Jaderberg Max","year":"2015","unstructured":"Max Jaderberg , Karen Simonyan , Andrew Zisserman , and Koray Kavukcuoglu . 2015. Spatial transformer networks. arXiv preprint arXiv:1506.02025 ( 2015 ). Max Jaderberg, Karen Simonyan, Andrew Zisserman, and Koray Kavukcuoglu. 2015. Spatial transformer networks. arXiv preprint arXiv:1506.02025 (2015)."},{"key":"e_1_3_2_2_14_1","volume-title":"Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114","author":"Kingma Diederik P","year":"2013","unstructured":"Diederik P Kingma and Max Welling . 2013. Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114 ( 2013 ). Diederik P Kingma and Max Welling. 2013. Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114 (2013)."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2011.6126543"},{"key":"e_1_3_2_2_16_1","volume-title":"International conference on machine learning. PMLR, 3825--3834","author":"Li Huaiyu","year":"2019","unstructured":"Huaiyu Li , Weiming Dong , Xing Mei , Chongyang Ma , Feiyue Huang , and Bao-Gang Hu . 2019 . LGM-Net: Learning to generate matching networks for few-shot learning . In International conference on machine learning. PMLR, 3825--3834 . Huaiyu Li, Weiming Dong, Xing Mei, Chongyang Ma, Feiyue Huang, and Bao-Gang Hu. 2019. LGM-Net: Learning to generate matching networks for few-shot learning. In International conference on machine learning. PMLR, 3825--3834."},{"key":"e_1_3_2_2_17_1","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","volume":"36","author":"Fei Mengjuan","year":"2022","unstructured":"Shuyuan Li, Huabin Liu, Rui Qian, Yuxi Li, John See, Mengjuan Fei , Xiaoyuan Yu , and Weiyao Lin . 2022 . TA2N: Two-Stage Action Alignment Network for Few-Shot Action Recognition . In Proceedings of the AAAI Conference on Artificial Intelligence , Vol. 36 . 1404--1411. Shuyuan Li, Huabin Liu, Rui Qian, Yuxi Li, John See, Mengjuan Fei, Xiaoyuan Yu, and Weiyao Lin. 2022. TA2N: Two-Stage Action Alignment Network for Few-Shot Action Recognition. In Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 36. 1404--1411."},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/MMUL.2020.2990863"},{"key":"e_1_3_2_2_19_1","volume-title":"Human in events: A large-scale benchmark for human-centric video analysis in complex events. arXiv preprint arXiv:2005.04490","author":"Lin Weiyao","year":"2020","unstructured":"Weiyao Lin , Huabin Liu , Shizhan Liu , Yuxi Li , Rui Qian , Tao Wang , Ning Xu , Hongkai Xiong , Guo-Jun Qi , and Nicu Sebe . 2020b. Human in events: A large-scale benchmark for human-centric video analysis in complex events. arXiv preprint arXiv:2005.04490 ( 2020 ). Weiyao Lin, Huabin Liu, Shizhan Liu, Yuxi Li, Rui Qian, Tao Wang, Ning Xu, Hongkai Xiong, Guo-Jun Qi, and Nicu Sebe. 2020b. Human in events: A large-scale benchmark for human-centric video analysis in complex events. arXiv preprint arXiv:2005.04490 (2020)."},{"key":"e_1_3_2_2_20_1","volume-title":"Learning scale-consistent attention part network for fine-grained image recognition","author":"Lin Weiyao","year":"2021","unstructured":"Huabin Liu, Jianguo Li, Dian Li, John See, and Weiyao Lin . 2021. Learning scale-consistent attention part network for fine-grained image recognition . IEEE Transactions on Multimedia ( 2021 ). Huabin Liu, Jianguo Li, Dian Li, John See, and Weiyao Lin. 2021. Learning scale-consistent attention part network for fine-grained image recognition. IEEE Transactions on Multimedia (2021)."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01264-9_8"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58571-6_6"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2021.108411"},{"key":"e_1_3_2_2_24_1","volume-title":"Temporal-Relational CrossTransformers for Few-Shot Action Recognition. arXiv preprint arXiv:2101.06184","author":"Perrett Toby","year":"2021","unstructured":"Toby Perrett , Alessandro Masullo , Tilo Burghardt , Majid Mirmehdi , and Dima Damen . 2021. Temporal-Relational CrossTransformers for Few-Shot Action Recognition. arXiv preprint arXiv:2101.06184 ( 2021 ). Toby Perrett, Alessandro Masullo, Tilo Burghardt, Majid Mirmehdi, and Dima Damen. 2021. Temporal-Relational CrossTransformers for Few-Shot Action Recognition. arXiv preprint arXiv:2101.06184 (2021)."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.590"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01240-3_4"},{"key":"e_1_3_2_2_27_1","volume-title":"Prototypical networks for few-shot learning. arXiv preprint arXiv:1703.05175","author":"Snell Jake","year":"2017","unstructured":"Jake Snell , Kevin Swersky , and Richard S Zemel . 2017. Prototypical networks for few-shot learning. arXiv preprint arXiv:1703.05175 ( 2017 ). Jake Snell, Kevin Swersky, and Richard S Zemel. 2017. Prototypical networks for few-shot learning. arXiv preprint arXiv:1703.05175 (2017)."},{"key":"e_1_3_2_2_28_1","volume-title":"Amir Roshan Zamir, and Mubarak Shah","author":"Soomro Khurram","year":"2012","unstructured":"Khurram Soomro , Amir Roshan Zamir, and Mubarak Shah . 2012 . UCF101: A dataset of 101 human actions classes from videos in the wild. arXiv preprint arXiv:1212.0402 (2012). Khurram Soomro, Amir Roshan Zamir, and Mubarak Shah. 2012. UCF101: A dataset of 101 human actions classes from videos in the wild. arXiv preprint arXiv:1212.0402 (2012)."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00675"},{"key":"e_1_3_2_2_31_1","volume-title":"Attention is all you need. arXiv preprint arXiv:1706.03762","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani , Noam Shazeer , Niki Parmar , Jakob Uszkoreit , Llion Jones , Aidan N Gomez , Lukasz Kaiser , and Illia Polosukhin . 2017. Attention is all you need. arXiv preprint arXiv:1706.03762 ( 2017 ). Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, Lukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. arXiv preprint arXiv:1706.03762 (2017)."},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01594"},{"key":"e_1_3_2_2_34_1","first-page":"2432","article-title":"Glance and focus: a dynamic approach to reducing spatial redundancy in image classification","volume":"33","author":"Wang Yulin","year":"2020","unstructured":"Yulin Wang , Kangchen Lv , Rui Huang , Shiji Song , Le Yang , and Gao Huang . 2020 . Glance and focus: a dynamic approach to reducing spatial redundancy in image classification . Advances in Neural Information Processing Systems , Vol. 33 (2020), 2432 -- 2444 . Yulin Wang, Kangchen Lv, Rui Huang, Shiji Song, Le Yang, and Gao Huang. 2020. Glance and focus: a dynamic approach to reducing spatial redundancy in image classification. Advances in Neural Information Processing Systems , Vol. 33 (2020), 2432--2444.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_35_1","volume-title":"Adafocus v2: End-to-end training of spatial dynamic networks for video recognition. arXiv preprint arXiv:2112.14238","author":"Wang Yulin","year":"2021","unstructured":"Yulin Wang , Yang Yue , Yuanze Lin , Haojun Jiang , Zihang Lai , Victor Kulikov , Nikita Orlov , Humphrey Shi , and Gao Huang . 2021b. Adafocus v2: End-to-end training of spatial dynamic networks for video recognition. arXiv preprint arXiv:2112.14238 ( 2021 ). Yulin Wang, Yang Yue, Yuanze Lin, Haojun Jiang, Zihang Lai, Victor Kulikov, Nikita Orlov, Humphrey Shi, and Gao Huang. 2021b. Adafocus v2: End-to-end training of spatial dynamic networks for video recognition. arXiv preprint arXiv:2112.14238 (2021)."},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00632"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00137"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58558-7_31"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jvcir.2021.103224"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00515"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00154"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01234-2_46"},{"key":"e_1_3_2_2_43_1","first-page":"1","article-title":"Label independent memory for semi-supervised few-shot video classification","volume":"01","author":"Zhu Linchao","year":"2020","unstructured":"Linchao Zhu and Yi Yang . 2020 . Label independent memory for semi-supervised few-shot video classification . IEEE Annals of the History of Computing 01 (2020), 1 -- 1 . Linchao Zhu and Yi Yang. 2020. Label independent memory for semi-supervised few-shot video classification. IEEE Annals of the History of Computing 01 (2020), 1--1.","journal-title":"IEEE Annals of the History of Computing"}],"event":{"name":"MM '22: The 30th ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Lisboa Portugal","acronym":"MM '22"},"container-title":["Proceedings of the 30th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503161.3547938","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3503161.3547938","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:00:31Z","timestamp":1750186831000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503161.3547938"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,10]]},"references-count":43,"alternative-id":["10.1145\/3503161.3547938","10.1145\/3503161"],"URL":"https:\/\/doi.org\/10.1145\/3503161.3547938","relation":{},"subject":[],"published":{"date-parts":[[2022,10,10]]},"assertion":[{"value":"2022-10-10","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}