{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:25:52Z","timestamp":1765308352393,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":47,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755463","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T05:47:42Z","timestamp":1761371262000},"page":"5726-5734","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Robust Understanding of Human-robot Social Interactions through Multimodal Distillation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5944-4157","authenticated-orcid":false,"given":"Tongfei","family":"Bian","sequence":"first","affiliation":[{"name":"School of Computer Science, University of Glasgow, Glasgow, United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9858-6844","authenticated-orcid":false,"given":"Mathieu","family":"Chollet","sequence":"additional","affiliation":[{"name":"School of Computer Science, University of Glasgow, Glasgow, United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2167-4891","authenticated-orcid":false,"given":"Tanaya","family":"Guha","sequence":"additional","affiliation":[{"name":"School of Computer Science, University of Glasgow, Glasgow, United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.robot.2023.104568"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV48630.2021.00317"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/s12369-020-00720-2"},{"volume-title":"Exploring 3D Human Pose Estimation and Forecasting from the Robot's Perspective: The HARPER Dataset","author":"Avogaro Andrea","key":"e_1_3_2_1_4_1","unstructured":"Andrea Avogaro, Andrea Toaiari, Federico Cunico, Xiangmin Xu, Haralambos Dafas, Alessandro Vinciarelli, Emma Li, and Marco Cristani. 2024. Exploring 3D Human Pose Estimation and Forecasting from the Robot's Perspective: The HARPER Dataset. In Prco IROS. IEEE, 5828--5835."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1007\/s12369-019-00591-2"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICME59968.2025.11210231"},{"key":"e_1_3_2_1_7_1","volume-title":"A blessing or a burden? Exploring worker perspectives of using a social robot in a church. arXiv preprint arXiv:2507.22903","author":"Blair Andrew","year":"2025","unstructured":"Andrew Blair, Peggy Gregory, and Mary Ellen Foster. 2025. A blessing or a burden? Exploring worker perspectives of using a social robot in a church. arXiv preprint arXiv:2507.22903 (2025)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1455"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3448018.3458008"},{"key":"e_1_3_2_1_10_1","volume-title":"A systematic review of robustness in deep learning for computer vision: Mind the gap? arXiv preprint arXiv:2112.00639","author":"Drenkow Nathan","year":"2021","unstructured":"Nathan Drenkow, Numair Sani, Ilya Shpitser, and Mathias Unberath. 2021. A systematic review of robustness in deep learning for computer vision: Mind the gap? arXiv preprint arXiv:2112.00639 (2021)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00298"},{"key":"e_1_3_2_1_12_1","volume-title":"Alphapose: Whole-body regional multi-person pose estimation and tracking in real-time","author":"Fang Hao-Shu","year":"2022","unstructured":"Hao-Shu Fang, Jiefeng Li, Hongyang Tang, Chao Xu, Haoyi Zhu, Yuliang Xiu, Yong-Lu Li, and Cewu Lu. 2022. Alphapose: Whole-body regional multi-person pose estimation and tracking in real-time. IEEE transactions on pattern analysis and machine intelligence 45, 6 (2022), 7157--7173."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1007\/s12369-020-00661-w"},{"key":"e_1_3_2_1_14_1","volume-title":"MuMMER: Socially Intelligent Human-Robot Interaction in Public Spaces. In Artificial Intelligence for Human-Robot Interaction Symposium (AI-HRI).","author":"Foster Mary Ellen","year":"2019","unstructured":"Mary Ellen Foster, Bart Craenen, Amol Deshmukh, Oliver Lemon, Emanuele Bastianelli, Christian Dondrup, Ioannis Papaioannou, Andrea Vanzo, Jean-Marc Odobez, Olivier Can\u00e9vet, et al. 2019. MuMMER: Socially Intelligent Human-Robot Interaction in Public Spaces. In Artificial Intelligence for Human-Robot Interaction Symposium (AI-HRI)."},{"key":"e_1_3_2_1_15_1","volume-title":"Framewise phoneme classification with bidirectional LSTM networks","author":"Graves Alex","year":"2047","unstructured":"Alex Graves and J\u00fcrgen Schmidhuber. 2005. Framewise phoneme classification with bidirectional LSTM networks, Vol. 4. IEEE, 2047--2052."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2023.3332569"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_18_1","volume-title":"Distilling the Knowledge in a Neural Network. stat 1050","author":"Hinton Geoffrey","year":"2015","unstructured":"Geoffrey Hinton, Oriol Vinyals, and Jeff Dean. 2015. Distilling the Knowledge in a Neural Network. stat 1050 (2015), 9."},{"key":"e_1_3_2_1_19_1","unstructured":"Chaudhary Muhammad Aqdus Ilyas Rita Nunes and Thomas B Moeslund. 2021. Deep Emotion Recognition through Upper Body Movements and Facial Expression.. In VISIGRAPP (5: VISAPP). 669--679."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS45743.2020.9340987"},{"key":"e_1_3_2_1_21_1","unstructured":"Will Kay Joao Carreira Karen Simonyan Brian Zhang Chloe Hillier Sudheendra Vijayanarasimhan Fabio Viola Tim Green Trevor Back Paul Natsev et al. 2017. The kinetics human action video dataset. arXiv preprint arXiv:1705.06950 (2017)."},{"key":"e_1_3_2_1_22_1","volume-title":"Proc ICLR.","author":"Kingma Diederik P","year":"2015","unstructured":"Diederik P Kingma and Jimmy Ba. 2015. Adam: A method for stochastic optimization. In Proc ICLR."},{"key":"e_1_3_2_1_23_1","volume-title":"Kipf and Max Welling","author":"Thomas","year":"2017","unstructured":"Thomas N. Kipf and Max Welling. 2017. Semi-Supervised Classification with Graph Convolutional Networks. In Proc ICLR."},{"key":"e_1_3_2_1_24_1","volume-title":"Austin Reiter, and Gregory D Hager","author":"Lea Colin","year":"2017","unstructured":"Colin Lea, Michael D Flynn, Rene Vidal, Austin Reiter, and Gregory D Hager. 2017. Temporal convolutional networks for action segmentation and detection. In proc CVPR. 156--165."},{"key":"e_1_3_2_1_25_1","volume-title":"Attention-Based Multimodal Fusion for Estimating Human Emotion in Real-World HRI. In Companion of HRI (HRI '20)","author":"Li Yuanchao","year":"2020","unstructured":"Yuanchao Li, Tianyu Zhao, and Xun Shen. 2020. Attention-Based Multimodal Fusion for Estimating Human Emotion in Real-World HRI. In Companion of HRI (HRI '20). Association for Computing Machinery, 340--342."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3656580"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2018.2884793"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00022"},{"key":"e_1_3_2_1_29_1","volume-title":"Representation learning with contrastive predictive coding. arXiv preprint arXiv:1807.03748","author":"van den Oord Aaron","year":"2018","unstructured":"Aaron van den Oord, Yazhe Li, and Oriol Vinyals. 2018. Representation learning with contrastive predictive coding. arXiv preprint arXiv:1807.03748 (2018)."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3526109"},{"key":"e_1_3_2_1_31_1","volume-title":"ICPR international workshops and challenges: virtual event, January 10--15, 2021, Proceedings, Part III. Springer, 694--701","author":"Plizzari Chiara","year":"2021","unstructured":"Chiara Plizzari, Marco Cannici, and Matteo Matteucci. 2021. Spatial temporal transformer network for skeleton-based action recognition. In Pattern recognition. ICPR international workshops and challenges: virtual event, January 10--15, 2021, Proceedings, Part III. Springer, 694--701."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"crossref","unstructured":"Shokoofeh Pourmehr Jack Thomas Jake Bruce Jens Wawerla and Richard Vaughan. 2017. Robust sensor fusion for finding HRI partners in a crowd. In Prco ICRA. 3272--3278.","DOI":"10.1109\/ICRA.2017.7989373"},{"key":"e_1_3_2_1_33_1","volume-title":"Antoine Chassang, Carlo Gatta, and Yoshua Bengio.","author":"Romero Adriana","year":"2014","unstructured":"Adriana Romero, Nicolas Ballas, Samira Ebrahimi Kahou, Antoine Chassang, Carlo Gatta, and Yoshua Bengio. 2014. Fitnets: Hints for thin deep nets. arXiv preprint arXiv:1412.6550 (2014)."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/2696454.2696462"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2013.352"},{"key":"e_1_3_2_1_36_1","volume-title":"Proc ICML. PMLR, 1139--1147","author":"Sutskever Ilya","year":"2013","unstructured":"Ilya Sutskever, James Martens, George Dahl, and Geoffrey Hinton. 2013. On the importance of initialization and momentum in deep learning. In Proc ICML. PMLR, 1139--1147."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3570169"},{"key":"e_1_3_2_1_38_1","volume-title":"Attention is all you need. Advances in neural information processing systems 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_39_1","volume-title":"Proc ICLR.","author":"Veli\u010dkovi\u0107 Petar","year":"2018","unstructured":"Petar Veli\u010dkovi\u0107, Guillem Cucurull, Arantxa Casanova, Adriana Romero, Pietro Li\u00f2, and Yoshua Bengio. 2018. Graph Attention Networks. In Proc ICLR."},{"key":"e_1_3_2_1_40_1","volume-title":"Kdgan: Knowledge distillation with generative adversarial networks. Advances in neural information processing systems 31","author":"Wang Xiaojie","year":"2018","unstructured":"Xiaojie Wang, Rui Zhang, Yu Sun, and Jianzhong Qi. 2018. Kdgan: Knowledge distillation with generative adversarial networks. Advances in neural information processing systems 31 (2018)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-24667-8_23"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.12328"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626954"},{"volume-title":"Context aware human-robot and human-agent interaction","author":"Yumak Zerrin","key":"e_1_3_2_1_44_1","unstructured":"Zerrin Yumak and Nadia Magnenat-Thalmann. 2015. Multimodal and multi-party social interactions. In Context aware human-robot and human-agent interaction. Springer, 275--298."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ROBIO.2018.8665127"},{"key":"e_1_3_2_1_46_1","volume-title":"Proc ICLR.","author":"Zagoruyko Sergey","year":"2017","unstructured":"Sergey Zagoruyko and Nikos Komodakis. 2017. Paying More Attention to Attention: Improving the Performance of Convolutional Neural Networks via Attention Transfer. In Proc ICLR."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1038\/s42256-022-00542-z"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755463","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:21:22Z","timestamp":1765308082000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755463"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":47,"alternative-id":["10.1145\/3746027.3755463","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755463","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}