{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T22:24:26Z","timestamp":1784413466716,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":53,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3612192","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:26:54Z","timestamp":1698391614000},"page":"2243-2251","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":20,"title":["On the Importance of Spatial Relations for Few-shot Action Recognition"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-8471-6983","authenticated-orcid":false,"given":"Yilun","family":"Zhang","sequence":"first","affiliation":[{"name":"Shanghai Key Lab of Intel. Info. Proc., School of CS, Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0412-5500","authenticated-orcid":false,"given":"Yuqian","family":"Fu","sequence":"additional","affiliation":[{"name":"Shanghai Key Lab of Intel. Info. Proc., School of CS, Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2099-4973","authenticated-orcid":false,"given":"Xingjun","family":"Ma","sequence":"additional","affiliation":[{"name":"Shanghai Key Lab of Intel. Info. Proc., School of CS, Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4348-1559","authenticated-orcid":false,"given":"Lizhe","family":"Qi","sequence":"additional","affiliation":[{"name":"Academy for Engineering and Technology, Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3148-264X","authenticated-orcid":false,"given":"Jingjing","family":"Chen","sequence":"additional","affiliation":[{"name":"Shanghai Key Lab of Intel. Info. Proc., School of CS, Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8689-5807","authenticated-orcid":false,"given":"Zuxuan","family":"Wu","sequence":"additional","affiliation":[{"name":"Shanghai Key Lab of Intel. Info. Proc., School of CS, Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1907-8567","authenticated-orcid":false,"given":"Yu-Gang","family":"Jiang","sequence":"additional","affiliation":[{"name":"Shanghai Key Lab of Intel. Info. Proc., School of CS, Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Jamie Ryan Kiros, and Geoffrey E Hinton","author":"Ba Jimmy Lei","year":"2016","unstructured":"Jimmy Lei Ba, Jamie Ryan Kiros, and Geoffrey E Hinton. 2016. Layer normalization. arXiv preprint arXiv:1607.06450 (2016)."},{"key":"e_1_3_2_1_2_1","volume-title":"Tarn: Temporal attentive relation network for few-shot and zero-shot action recognition. arXiv preprint arXiv:1907.09021","author":"Bishay Mina","year":"2019","unstructured":"Mina Bishay, Georgios Zoumpourlis, and Ioannis Patras. 2019. Tarn: Temporal attentive relation network for few-shot and zero-shot action recognition. arXiv preprint arXiv:1907.09021 (2019)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Kaidi Cao Jingwei Ji Zhangjie Cao Chien-Yi Chang and Juan Carlos Niebles. 2020. Few-shot video classification via temporal alignment. In CVPR.","DOI":"10.1109\/CVPR42600.2020.01063"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","unstructured":"Mathilde Caron Hugo Touvron Ishan Misra Herv\u00e9 J\u00e9gou Julien Mairal Piotr Bojanowski and Armand Joulin. 2021. Emerging properties in self-supervised vision transformers. In ICCV.","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"Joao Carreira and Andrew Zisserman. 2017. Quo vadis action recognition? a new model and the kinetics dataset. In CVPR.","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_1_6_1","volume-title":"Conditional positional encodings for vision transformers. arXiv preprint arXiv:2102.10882","author":"Chu Xiangxiang","year":"2021","unstructured":"Xiangxiang Chu, Zhi Tian, Bo Zhang, Xinlong Wang, Xiaolin Wei, Huaxia Xia, and Chunhua Shen. 2021. Conditional positional encodings for vision transformers. arXiv preprint arXiv:2102.10882 (2021)."},{"key":"e_1_3_2_1_7_1","volume-title":"Imagenet: A large-scale hierarchical image database. In CVPR.","author":"Deng Jia","year":"2009","unstructured":"Jia Deng, Wei Dong, Richard Socher, Li-Jia Li, Kai Li, and Li Fei-Fei. 2009. Imagenet: A large-scale hierarchical image database. In CVPR."},{"key":"e_1_3_2_1_8_1","volume-title":"Crosstransformers: spatially-aware few-shot transfer. NeurIPS","author":"Doersch Carl","year":"2020","unstructured":"Carl Doersch, Ankush Gupta, and Andrew Zisserman. 2020. Crosstransformers: spatially-aware few-shot transfer. NeurIPS (2020)."},{"key":"e_1_3_2_1_9_1","unstructured":"Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly et al. 2021. An image is worth 16x16 words: Transformers for image recognition at scale. In ICLR."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"crossref","unstructured":"Christoph Feichtenhofer Haoqi Fan Jitendra Malik and Kaiming He. 2019. Slowfast networks for video recognition. In ICCV.","DOI":"10.1109\/ICCV.2019.00630"},{"key":"e_1_3_2_1_11_1","unstructured":"Chelsea Finn Pieter Abbeel and Sergey Levine. 2017. Model-agnostic meta-learning for fast adaptation of deep networks. In ICML."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475655"},{"key":"e_1_3_2_1_13_1","unstructured":"Yuqian Fu Chengrong Wang Yanwei Fu Yu-Xiong Wang Cong Bai Xiangyang Xue and Yu-Gang Jiang. 2019. Embodied one-shot video recognition: Learning from actions of a virtual embodied agent. In ACM Multimedia."},{"key":"e_1_3_2_1_14_1","unstructured":"Yuqian Fu Yu Xie Yanwei Fu and Yu-Gang Jiang. [n. d.]. StyleAdv: Meta Style Adversarial Training for Cross-Domain Few-Shot Learning. In CVPR."},{"key":"e_1_3_2_1_15_1","unstructured":"Yuqian Fu Li Zhang Junke Wang Yanwei Fu and Yu-Gang Jiang. 2020. Depth guided adaptive meta-fusion network for few-shot video recognition. In ACM Multimedia."},{"key":"e_1_3_2_1_16_1","volume-title":"Vincent Michalski, Joanna Materzynska, Susanne Westphal, Heuna Kim, Valentin Haenel, Ingo Fruend, Peter Yianilos, Moritz Mueller-Freitag, et al.","author":"Goyal Raghav","year":"2017","unstructured":"Raghav Goyal, Samira Ebrahimi Kahou, Vincent Michalski, Joanna Materzynska, Susanne Westphal, Heuna Kim, Valentin Haenel, Ingo Fruend, Peter Yianilos, Moritz Mueller-Freitag, et al. 2017. The\" something something\" video database for learning and evaluating visual common sense. In ICCV."},{"key":"e_1_3_2_1_17_1","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep residual learning for image recognition. In CVPR."},{"key":"e_1_3_2_1_18_1","volume-title":"Cross attention network for few-shot classification. NeurIPS","author":"Hou Ruibing","year":"2019","unstructured":"Ruibing Hou, Hong Chang, Bingpeng Ma, Shiguang Shan, and Xilin Chen. 2019. Cross attention network for few-shot classification. NeurIPS (2019)."},{"key":"e_1_3_2_1_19_1","unstructured":"Shell Xu Hu Da Li Jan St\u00fchmer Minyoung Kim and Timothy M Hospedales. 2022. Pushing the Limits of Simple Pipelines for Few-Shot Learning: External Data and Fine-Tuning Make a Difference. In CVPR."},{"key":"e_1_3_2_1_20_1","unstructured":"Gregory Koch Richard Zemel Ruslan Salakhutdinov et al. 2015. Siamese neural networks for one-shot image recognition. In ICML deep learning workshop."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"crossref","unstructured":"Hildegard Kuehne Hueihan Jhuang Est\u00edbaliz Garrote Tomaso Poggio and Thomas Serre. 2011. HMDB: a large video database for human motion recognition. In ICCV.","DOI":"10.1109\/ICCV.2011.6126543"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2019.00166"},{"key":"e_1_3_2_1_23_1","volume-title":"Tsm: Temporal shift module for efficient video understanding. In ICCV.","author":"Lin Ji","year":"2019","unstructured":"Ji Lin, Chuang Gan, and Song Han. 2019. Tsm: Temporal shift module for efficient video understanding. In ICCV."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"crossref","unstructured":"Ze Liu Yutong Lin Yue Cao Han Hu Yixuan Wei Zheng Zhang Stephen Lin and Baining Guo. 2021. Swin transformer: Hierarchical vision transformer using shifted windows. In ICCV.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_1_25_1","unstructured":"Tsendsuren Munkhdalai and Hong Yu. 2017. Meta networks. In ICML."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"crossref","unstructured":"Toby Perrett Alessandro Masullo Tilo Burghardt Majid Mirmehdi and Dima Damen. 2021. Temporal-relational crosstransformers for few-shot action recognition. In CVPR.","DOI":"10.1109\/CVPR46437.2021.00054"},{"key":"e_1_3_2_1_27_1","unstructured":"Zhaofan Qiu Ting Yao and Tao Mei. 2017. Learning spatio-temporal representation with pseudo-3d residual networks. In ICCV."},{"key":"e_1_3_2_1_28_1","volume-title":"Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al.","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In ICML."},{"key":"e_1_3_2_1_29_1","unstructured":"Sachin Ravi and Hugo Larochelle. 2016. Optimization as a model for few-shot learning. (2016)."},{"key":"e_1_3_2_1_30_1","unstructured":"Adam Santoro Sergey Bartunov Matthew Botvinick Daan Wierstra and Timothy Lillicrap. 2016. Meta-learning with memory-augmented neural networks. In ICML."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"crossref","unstructured":"Laura Sevilla-Lara Shengxin Zha Zhicheng Yan Vedanuj Goswami Matt Feiszli and Lorenzo Torresani. 2021. Only time can tell: Discovering temporal data for temporal modeling. In WACV.","DOI":"10.1109\/WACV48630.2021.00058"},{"key":"e_1_3_2_1_32_1","volume-title":"Prototypical networks for few-shot learning. NeurIPS","author":"Snell Jake","year":"2017","unstructured":"Jake Snell, Kevin Swersky, and Richard Zemel. 2017. Prototypical networks for few-shot learning. NeurIPS (2017)."},{"key":"e_1_3_2_1_33_1","volume-title":"Amir Roshan Zamir, and Mubarak Shah","author":"Soomro Khurram","year":"2012","unstructured":"Khurram Soomro, Amir Roshan Zamir, and Mubarak Shah. 2012. UCF101: A dataset of 101 human actions classes from videos in the wild. arXiv preprint arXiv:1212.0402 (2012)."},{"key":"e_1_3_2_1_34_1","volume-title":"Philip HS Torr, and Timothy M Hospedales","author":"Sung Flood","year":"2018","unstructured":"Flood Sung, Yongxin Yang, Li Zhang, Tao Xiang, Philip HS Torr, and Timothy M Hospedales. 2018. Learning to compare: Relation network for few-shot learning. In CVPR."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"crossref","unstructured":"Hao Tang Zechao Li Zhimao Peng and Jinhui Tang. 2020. Blockmix: meta regularization and self-calibrated inference for metric-based meta-learning. In ACM Multimedia.","DOI":"10.1145\/3394171.3413884"},{"key":"e_1_3_2_1_36_1","volume-title":"Learning attention-guided pyramidal features for few-shot fine-grained recognition. Pattern Recognition","author":"Tang Hao","year":"2022","unstructured":"Hao Tang, Chengcheng Yuan, Zechao Li, and Jinhui Tang. 2022. Learning attention-guided pyramidal features for few-shot fine-grained recognition. Pattern Recognition (2022)."},{"key":"e_1_3_2_1_37_1","volume-title":"Fahad Shahbaz Khan, and Bernard Ghanem.","author":"Thatipelli Anirudh","year":"2022","unstructured":"Anirudh Thatipelli, Sanath Narayan, Salman Khan, Rao Muhammad Anwer, Fahad Shahbaz Khan, and Bernard Ghanem. 2022. Spatio-temporal relation modeling for few-shot action recognition. In CVPR."},{"key":"e_1_3_2_1_38_1","volume-title":"Mlp-mixer: An all-mlp architecture for vision. NeurIPS","author":"Tolstikhin Ilya O","year":"2021","unstructured":"Ilya O Tolstikhin, Neil Houlsby, Alexander Kolesnikov, Lucas Beyer, Xiaohua Zhai, Thomas Unterthiner, Jessica Yung, Andreas Steiner, Daniel Keysers, Jakob Uszkoreit, et al. 2021. Mlp-mixer: An all-mlp architecture for vision. NeurIPS (2021)."},{"key":"e_1_3_2_1_39_1","unstructured":"Hugo Touvron Matthieu Cord Matthijs Douze Francisco Massa Alexandre Sablayrolles and Herv\u00e9 J\u00e9gou. 2021. Training data-efficient image transformers & distillation through attention. In ICML."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"crossref","unstructured":"Du Tran Lubomir Bourdev Rob Fergus Lorenzo Torresani and Manohar Paluri. 2015. Learning spatiotemporal features with 3d convolutional networks. In ICCV.","DOI":"10.1109\/ICCV.2015.510"},{"key":"e_1_3_2_1_41_1","volume-title":"Attention is all you need. NeurIPS","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. NeurIPS (2017)."},{"key":"e_1_3_2_1_42_1","unstructured":"Oriol Vinyals Charles Blundell Timothy Lillicrap Daan Wierstra et al. 2016. Matching networks for one shot learning. NeurIPS (2016)."},{"key":"e_1_3_2_1_43_1","volume-title":"Strata: Self-training with task augmentation for better few-shot learning. arXiv preprint arXiv:2109.06270","author":"Vu Tu","year":"2021","unstructured":"Tu Vu, Minh-Thang Luong, Quoc V Le, Grady Simon, and Mohit Iyyer. 2021. Strata: Self-training with task augmentation for better few-shot learning. arXiv preprint arXiv:2109.06270 (2021)."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"crossref","unstructured":"Xiang Wang Shiwei Zhang Zhiwu Qing Mingqian Tang Zhengrong Zuo Changxin Gao Rong Jin and Nong Sang. 2022. Hybrid relation guided set matching for few-shot action recognition. In CVPR.","DOI":"10.1109\/CVPR52688.2022.01932"},{"key":"e_1_3_2_1_45_1","unstructured":"Jiamin Wu Tianzhu Zhang Zhe Zhang Feng Wu and Yongdong Zhang. 2022. Motion-modulated temporal fragment alignment network for few-shot action recognition. In CVPR."},{"key":"e_1_3_2_1_46_1","volume-title":"Philip HS Torr, and Piotr Koniusz","author":"Zhang Hongguang","year":"2020","unstructured":"Hongguang Zhang, Li Zhang, Xiaojuan Qi, Hongdong Li, Philip HS Torr, and Piotr Koniusz. 2020. Few-shot action recognition with permutation-invariant attention. In ECCV."},{"key":"e_1_3_2_1_47_1","volume-title":"DETA: Denoised Task Adaptation for Few-Shot Learning. arXiv preprint arXiv:2303.06315","author":"Zhang Ji","year":"2023","unstructured":"Ji Zhang, Lianli Gao, Xu Luo, Hengtao Shen, and Jingkuan Song. 2023. DETA: Denoised Task Adaptation for Few-Shot Learning. arXiv preprint arXiv:2303.06315 (2023)."},{"key":"e_1_3_2_1_48_1","volume-title":"Progressive meta-learning with curriculum. TCSVT","author":"Zhang Ji","year":"2022","unstructured":"Ji Zhang, Jingkuan Song, Lianli Gao, Ye Liu, and Heng Tao Shen. 2022. Progressive meta-learning with curriculum. TCSVT (2022)."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"crossref","unstructured":"Xueting Zhang Debin Meng Henry Gouk and Timothy M Hospedales. 2021. Shallow bayesian meta learning for real-world few-shot recognition. In ICCV.","DOI":"10.1109\/ICCV48922.2021.00069"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"crossref","unstructured":"Yizhou Zhao Xun Guo and Yan Lu. 2022. Semantic-aligned Fusion Transformer for One-shot Object Detection. In CVPR.","DOI":"10.1109\/CVPR52688.2022.00745"},{"key":"e_1_3_2_1_51_1","volume-title":"Hypertransformer: Model generation for supervised and semi-supervised few-shot learning. In ICML.","author":"Zhmoginov Andrey","year":"2022","unstructured":"Andrey Zhmoginov, Mark Sandler, and Maksym Vladymyrov. 2022. Hypertransformer: Model generation for supervised and semi-supervised few-shot learning. In ICML."},{"key":"e_1_3_2_1_52_1","unstructured":"Linchao Zhu and Yi Yang. 2018. Compound memory networks for few-shot video classification. In ECCV."},{"key":"e_1_3_2_1_53_1","volume-title":"Tgdm: Target guided dynamic mixup for cross-domain few-shot learning. In ACM Multimedia.","author":"Zhuo Linhai","year":"2022","unstructured":"Linhai Zhuo, Yuqian Fu, Jingjing Chen, Yixin Cao, and Yu-Gang Jiang. 2022. Tgdm: Target guided dynamic mixup for cross-domain few-shot learning. In ACM Multimedia."}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada","acronym":"MM '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612192","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3612192","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:01:54Z","timestamp":1755820914000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612192"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":53,"alternative-id":["10.1145\/3581783.3612192","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3612192","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}