{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,16]],"date-time":"2026-05-16T16:11:39Z","timestamp":1778947899390,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":81,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"the National Key R&D Program of China","award":["No. 2018AAA0102001"],"award-info":[{"award-number":["No. 2018AAA0102001"]}]},{"name":"the China Postdoctoral Science Foundation","award":["Grant No. 2023M731596"],"award-info":[{"award-number":["Grant No. 2023M731596"]}]},{"name":"the Natural Science Foundation of Jiangsu Province","award":["Grant No. BK20211520"],"award-info":[{"award-number":["Grant No. BK20211520"]}]},{"name":"the National Natural Science Foundation of China","award":["Grant Nos. 62222207, 62072245, 62276134, and 62002160"],"award-info":[{"award-number":["Grant Nos. 62222207, 62072245, 62276134, and 62002160"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3612435","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:30Z","timestamp":1698391650000},"page":"7666-7675","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":10,"title":["MUP: Multi-granularity Unified Perception for Panoramic Activity Recognition"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-2624-1234","authenticated-orcid":false,"given":"Meiqi","family":"Cao","sequence":"first","affiliation":[{"name":"Nanjing University of Science and Technology, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0694-9458","authenticated-orcid":false,"given":"Rui","family":"Yan","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4902-4663","authenticated-orcid":false,"given":"Xiangbo","family":"Shu","sequence":"additional","affiliation":[{"name":"Nanjing University of Science and Technology, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-2637-5095","authenticated-orcid":false,"given":"Jiachao","family":"Zhang","sequence":"additional","affiliation":[{"name":"Nanjing Institute of Technology, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6127-9146","authenticated-orcid":false,"given":"Jinpeng","family":"Wang","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5487-9845","authenticated-orcid":false,"given":"Guo-Sen","family":"Xie","sequence":"additional","affiliation":[{"name":"Nanjing University of Science and Technology, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10599-4_37"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2013.171"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-33765-9_14"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00808"},{"key":"e_1_3_2_1_6_1","volume-title":"Jamie Ryan Kiros, and Geoffrey E Hinton","author":"Ba Jimmy Lei","year":"2016","unstructured":"Jimmy Lei Ba, Jamie Ryan Kiros, and Geoffrey E Hinton. 2016. Layer normalization. arXiv preprint arXiv:1607.06450 (2016)."},{"key":"e_1_3_2_1_7_1","volume-title":"Proceedings of the International Conference on Machine Learning","volume":"2","author":"Bertasius Gedas","year":"2021","unstructured":"Gedas Bertasius, Heng Wang, and Lorenzo Torresani. 2021. Is space-time attention all you need for video understanding?. In Proceedings of the International Conference on Machine Learning, Vol. 2. 4."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547822"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_1_11_1","volume-title":"Proceedings of the International Conference on Neural Information Processing Systems 34","author":"Cheng Bowen","year":"2021","unstructured":"Bowen Cheng, Alex Schwing, and Alexander Kirillov. 2021. Per-pixel classification is not all you need for semantic segmentation. Proceedings of the International Conference on Neural Information Processing Systems 34 (2021), 17864--17875."},{"key":"e_1_3_2_1_12_1","volume-title":"Proceedings of the IEEE International Conference on Computer Vision Workshop. 1282--1289","author":"Choi Wongun","year":"2009","unstructured":"Wongun Choi, Khuram Shahid, and Silvio Savarese. 2009. What are they doing?: Collective activity classification using spatio-temporal relationship among people. In Proceedings of the IEEE International Conference on Computer Vision Workshop. 1282--1289."},{"key":"e_1_3_2_1_13_1","unstructured":"Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xi-aohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly et al. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58545-7_11"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.02031"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.3390\/jimaging6090095"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00675"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00092"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"crossref","unstructured":"Michael A Goodrich Alan C Schultz et al. 2008. Human--robot interaction: a survey. Foundations and Trends\u00ae in Human-Computer Interaction 1 3 (2008) 203--275.","DOI":"10.1561\/1100000005"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00525"},{"key":"e_1_3_2_1_21_1","volume-title":"Panoramic Human Activity Recognition. In European Conference on Computer Vision. 244--261","author":"Han Ruize","year":"2022","unstructured":"Ruize Han, Haomin Yan, Jiacheng Li, Songmiao Wang, Wei Feng, and Song Wang. 2022. Panoramic Human Activity Recognition. In European Conference on Computer Vision. 244--261."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.322"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00106"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2018.2883522"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3505244"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01945"},{"key":"e_1_3_2_1_28_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980","author":"Kingma Diederik P","year":"2014","unstructured":"Diederik P Kingma and Jimmy Ba. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-022-01594-9"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/SURV.2012.110112.00192"},{"key":"e_1_3_2_1_31_1","volume-title":"Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805","author":"Lee JDMCK","year":"2018","unstructured":"JDMCK Lee and K Toutanova. 2018. Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3444685.3446255"},{"key":"e_1_3_2_1_33_1","volume-title":"Uniformer: Unifying convolution and self-attention for visual recognition. arXiv preprint arXiv:2201.09450","author":"Li Kunchang","year":"2022","unstructured":"Kunchang Li, Yali Wang, Junhao Zhang, Peng Gao, Guanglu Song, Yu Liu, Hongsheng Li, and Yu Qiao. 2022. Uniformer: Unifying convolution and self-attention for visual recognition. arXiv preprint arXiv:2201.09450 (2022)."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01341"},{"key":"e_1_3_2_1_35_1","volume-title":"Improved multiscale vision transformers for classification and detection. arXiv preprint arXiv:2112.01526","author":"Li Yanghao","year":"2021","unstructured":"Yanghao Li, Chao-Yuan Wu, Haoqi Fan, Karttikeya Mangalam, Bo Xiong, Jitendra Malik, and Christoph Feichtenhofer. 2021. Improved multiscale vision transformers for classification and detection. arXiv preprint arXiv:2112.01526 (2021)."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00476"},{"key":"e_1_3_2_1_37_1","volume-title":"IEEE International Symposium on Circuits and Systems (ISCAS). IEEE, 2737--2740","author":"Lin Weiyao","year":"2008","unstructured":"Weiyao Lin, Ming-Ting Sun, Radha Poovandran, and Zhengyou Zhang. 2008. Human activity recognition for video surveillance. In IEEE International Symposium on Circuits and Systems (ISCAS). IEEE, 2737--2740."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"e_1_3_2_1_39_1","volume-title":"Jrdb: A dataset and benchmark of egocentric robot visual perception of humans in built environments","author":"Martin-Martin Roberto","year":"2021","unstructured":"Roberto Martin-Martin, Mihir Patel, Hamid Rezatofighi, Abhijeet Shenoi, Jun-Young Gwak, Eric Frankel, Amir Sadeghian, and Silvio Savarese. 2021. Jrdb: A dataset and benchmark of egocentric robot visual perception of humans in built environments. IEEE Transactions on Pattern Analysis and Machine Intelligence (2021)."},{"key":"e_1_3_2_1_40_1","first-page":"12493","article-title":"Keeping your eye on the ball: Trajectory attention in video transformers","volume":"34","author":"Patrick Mandela","year":"2021","unstructured":"Mandela Patrick, Dylan Campbell, Yuki Asano, Ishan Misra, Florian Metze, Christoph Feichtenhofer, Andrea Vedaldi, and Jo\u00e3o F Henriques. 2021. Keeping your eye on the ball: Trajectory attention in video transformers. NeurIPS 34 (2021), 12493--12506.","journal-title":"NeurIPS"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_5"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/BigMM.2018.8499076"},{"key":"e_1_3_2_1_43_1","first-page":"1926","article-title":"Deep learning in sport video analysis: a review","volume":"18","author":"Rangasamy Keerthana","year":"2020","unstructured":"Keerthana Rangasamy, Muhammad Amir As'ari, Nur Azmina Rahmad, Nurul Fathiah Ghazali, and Saharudin Ismail. 2020. Deep learning in sport video analysis: a review. Telecommunication Computing Electronics and Control 18, 4 (2020), 1926--1933.","journal-title":"Telecommunication Computing Electronics and Control"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472168"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-010-0355-5"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2006.242"},{"key":"e_1_3_2_1_47_1","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 4576--4584","author":"Shu Tianmin","year":"2015","unstructured":"Tianmin Shu, Dan Xie, Brandon Rothrock, Sinisa Todorovic, and Song Chun Zhu. 2015. Joint inference of groups, events and human roles in aerial videos. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 4576--4584."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2019.2942030"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2020.101178"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2020.2996736"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00717"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3183112"},{"key":"e_1_3_2_1_53_1","volume-title":"Hunting Group Clues with Transformers for Social Group Activity Recognition. In European Conference on Computer Vision. 19--35","author":"Tamura Masato","year":"2022","unstructured":"Masato Tamura, Rahul Vishwakarma, and Ravigopal Vennelakanti. 2022. Hunting Group Clues with Transformers for Social Group Activity Recognition. In European Conference on Computer Vision. 19--35."},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1049\/ip-vis:20041147"},{"key":"e_1_3_2_1_56_1","volume-title":"Proceedings of the International Conference on Neural Information Processing Systems 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Proceedings of the International Conference on Neural Information Processing Systems 30."},{"key":"e_1_3_2_1_57_1","volume-title":"Graph attention networks. arXiv preprint arXiv:1710.10903","author":"Veli\u010dkovi\u0107 Petar","year":"2017","unstructured":"Petar Veli\u010dkovi\u0107, Guillem Cucurull, Arantxa Casanova, Adriana Romero, Pietro Lio, and Yoshua Bengio. 2017. Graph attention networks. arXiv preprint arXiv:1710.10903 (2017)."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2013.441"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"e_1_3_2_1_60_1","volume-title":"Learning deep transformer models for machine translation. arXiv preprint arXiv:1906.01787","author":"Wang Qiang","year":"2019","unstructured":"Qiang Wang, Bei Li, Tong Xiao, Jingbo Zhu, Changliang Li, Derek F Wong, and Lidia S Chao. 2019. Learning deep transformer models for machine translation. arXiv preprint arXiv:1906.01787 (2019)."},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475253"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00333"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01020"},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475201"},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01267-0_19"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3416284"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01760"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547862"},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2021.3085567"},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240508.3240572"},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3261659"},{"key":"e_1_3_2_1_72_1","volume-title":"HiGCIN: Hierarchical graph-based cross inference network for group activity recognition","author":"Yan Rui","year":"2020","unstructured":"Rui Yan, Lingxi Xie, Jinhui Tang, Xiangbo Shu, and Qi Tian. 2020. HiGCIN: Hierarchical graph-based cross inference network for group activity recognition. IEEE Transactions on Pattern Analysis and Machine Intelligence (2020)."},{"key":"e_1_3_2_1_73_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i4.16437"},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00738"},{"key":"e_1_3_2_1_75_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3416301"},{"key":"e_1_3_2_1_76_1","volume-title":"Pro-ceedings of the International Conference on Neural Information Processing Systems 17","author":"Zelnik-Manor Lihi","year":"2004","unstructured":"Lihi Zelnik-Manor and Pietro Perona. 2004. Self-tuning spectral clustering. Pro-ceedings of the International Conference on Neural Information Processing Systems 17 (2004)."},{"key":"e_1_3_2_1_77_1","doi-asserted-by":"publisher","DOI":"10.3390\/s19051005"},{"key":"e_1_3_2_1_78_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ymssp.2018.05.050"},{"key":"e_1_3_2_1_79_1","volume-title":"MLST-Former: multi-level spatial-temporal transformer for group activity recognition","author":"Zhu Xiaolin","year":"2022","unstructured":"Xiaolin Zhu, Yan Zhou, Dongli Wang, Wanli Ouyang, and Rui Su. 2022. MLST-Former: multi-level spatial-temporal transformer for group activity recognition. IEEE Transactions on Circuits and Systems for Video Technology (2022)."},{"key":"e_1_3_2_1_80_1","volume-title":"A comprehensive study of deep video action recognition. arXiv preprint arXiv:2012.06567","author":"Zhu Yi","year":"2020","unstructured":"Yi Zhu, Xinyu Li, Chunhui Liu, Mohammadreza Zolfaghari, Yuanjun Xiong, Chongruo Wu, Zhi Zhang, Joseph Tighe, R Manmatha, and Mu Li. 2020. A comprehensive study of deep video action recognition. arXiv preprint arXiv:2012.06567 (2020)."},{"key":"e_1_3_2_1_81_1","volume-title":"Proceedings of IAGG World Congress of Gerontology and Geriatrics","author":"Zouba Nadia","year":"2009","unstructured":"Nadia Zouba, Fran\u00e7ois Bremond, Monique Thonnat, Alain Anfosso, \u00c9ric Pascual, Patrick Mallea, V\u00e9ronique Mailland, and Olivier Guerin. 2009. Assessing computer systems for monitoring elderly people living at home. In Proceedings of IAGG World Congress of Gerontology and Geriatrics, Paris, France. 5--9."}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada","acronym":"MM '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612435","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3612435","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:04:35Z","timestamp":1755821075000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612435"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":81,"alternative-id":["10.1145\/3581783.3612435","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3612435","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}