{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T23:09:38Z","timestamp":1780355378896,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":25,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,3,31]],"date-time":"2025-03-31T00:00:00Z","timestamp":1743379200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"name":"Samsung Electronics America"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,3,31]]},"DOI":"10.1145\/3712678.3721885","type":"proceedings-article","created":{"date-parts":[[2025,4,3]],"date-time":"2025-04-03T06:20:32Z","timestamp":1743661232000},"page":"78-83","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["ModalityMirror: Enhancing Audio Classification in Modality Heterogeneity Federated Learning via Multimodal Distillation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2053-9068","authenticated-orcid":false,"given":"Tiantian","family":"Feng","sequence":"first","affiliation":[{"name":"University of Southern California, Los Angeles, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3676-0717","authenticated-orcid":false,"given":"Tuo","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of Southern California, Los Angeles, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3102-0867","authenticated-orcid":false,"given":"Salman","family":"Avestimehr","sequence":"additional","affiliation":[{"name":"University of Southern California, Los Angeles, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1052-6204","authenticated-orcid":false,"given":"Shrikanth","family":"Narayanan","sequence":"additional","affiliation":[{"name":"University of Southern California, Los Angeles, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,4,3]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Leveraging Foundation Models for Multi-modal Federated Learning with Incomplete Modality. In Joint European Conference on Machine Learning and Knowledge Discovery in Databases. Springer, 401--417","author":"Che Liwei","year":"2024","unstructured":"Liwei Che, Jiaqi Wang, Xinyue Liu, and Fenglong Ma. 2024. Leveraging Foundation Models for Multi-modal Federated Learning with Incomplete Modality. In Joint European Conference on Machine Learning and Knowledge Discovery in Databases. Springer, 401--417."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3534678.3539384"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_4_1","volume-title":"Proceedings of the 29th ACM SIGKDD Conference on Knowledge Discovery and Data Mining (2023","author":"Feng Tiantian","year":"1874","unstructured":"Tiantian Feng, Digbalay Bose, Tuo Zhang, Rajat Hebbar, Anil Ramakrishna, Rahul Gupta, Mi Zhang, Salman Avestimehr, and Shrikanth S. Narayanan. 2023. FedMultimodal: A Benchmark for Multimodal Federated Learning. Proceedings of the 29th ACM SIGKDD Conference on Knowledge Discovery and Data Mining (2023). https:\/\/api.semanticscholar.org\/CorpusID:259187495"},{"key":"e_1_3_2_1_5_1","volume-title":"Inverting gradients-how easy is it to break privacy in federated learning? Advances in neural information processing systems 33","author":"Geiping Jonas","year":"2020","unstructured":"Jonas Geiping, Hartmut Bauermeister, Hannah Dr\u00f6ge, and Michael Moeller. 2020. Inverting gradients-how easy is it to break privacy in federated learning? Advances in neural information processing systems 33 (2020), 16937--16947."},{"key":"e_1_3_2_1_6_1","volume-title":"Glass","author":"Gong Yuan","year":"2021","unstructured":"Yuan Gong, Cheng-I Lai, Yu-An Chung, and James R. Glass. 2021. SSAST: Self-Supervised Audio Spectrogram Transformer. ArXiv abs\/2110.09784 (2021). https:\/\/api.semanticscholar.org\/CorpusID:239024736"},{"key":"e_1_3_2_1_7_1","volume-title":"Glass","author":"Gong Yuan","year":"2022","unstructured":"Yuan Gong, Andrew Rouditchenko, Alexander H. Liu, David F. Harwath, Leonid Karlinsky, Hilde Kuehne, and James R. Glass. 2022. Contrastive Audio-Visual Masked Autoencoder. ArXiv abs\/2210.07839 (2022). https:\/\/api.semanticscholar.org\/CorpusID:252907593"},{"key":"e_1_3_2_1_8_1","volume-title":"Deep Residual Learning for Image Recognition. 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2015","author":"He Kaiming","year":"2015","unstructured":"Kaiming He, X. Zhang, Shaoqing Ren, and Jian Sun. 2015. Deep Residual Learning for Image Recognition. 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2015), 770--778. https:\/\/api.semanticscholar.org\/CorpusID:206594692"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"e_1_3_2_1_10_1","volume-title":"Multimodal machine learning in precision health: A scoping review. npj Digital Medicine 5, 1","author":"Kline Adrienne","year":"2022","unstructured":"Adrienne Kline, Hanyin Wang, Yikuan Li, Saya Dennis, Meghan Hutch, Zhenxing Xu, Fei Wang, Feixiong Cheng, and Yuan Luo. 2022. Multimodal machine learning in precision health: A scoping review. npj Digital Medicine 5, 1 (2022), 171."},{"key":"e_1_3_2_1_11_1","volume-title":"Multibench: Multi-scale benchmarks for multimodal representation learning. Advances in neural information processing systems","author":"Liang Paul Pu","year":"2021","unstructured":"Paul Pu Liang, Yiwei Lyu, Xiang Fan, Zetian Wu, Yun Cheng, Jason Wu, Leslie Chen, Peter Wu, Michelle A Lee, Yuke Zhu, et al. 2021. Multibench: Multi-scale benchmarks for multimodal representation learning. Advances in neural information processing systems 2021, DB1 (2021), 1."},{"key":"e_1_3_2_1_12_1","volume-title":"Effectiveness of VR-based training on improving occupants' response and preparedness for active shooter incidents. Safety science 164","author":"Liu Ruying","year":"2023","unstructured":"Ruying Liu, Burcin Becerik-Gerber, and Gale M Lucas. 2023. Effectiveness of VR-based training on improving occupants' response and preparedness for active shooter incidents. Safety science 164 (2023), 106175."},{"key":"e_1_3_2_1_13_1","volume-title":"Computing in Civil Engineering","author":"Liu Ruying","year":"2023","unstructured":"Ruying Liu, Bur\u00e7in Becerik-Gerber, Gale M Lucas, and Kelly Busta. [n. d.]. Development of a VR Training Platform for Active Shooter Incident Preparedness in Healthcare Environments via a Stakeholder-Engaged Process. In Computing in Civil Engineering 2023. 45--53."},{"key":"e_1_3_2_1_14_1","volume-title":"Enhancing Building Safety Design for Active Shooter Incidents: Exploration of Building Exit Parameters using Reinforcement Learning-Based Simulations. arXiv preprint arXiv:2407.10441","author":"Liu Ruying","year":"2024","unstructured":"Ruying Liu, Wanjing Wu, Burcin Becerik-Gerber, and Gale M Lucas. 2024. Enhancing Building Safety Design for Active Shooter Incidents: Exploration of Building Exit Parameters using Reinforcement Learning-Based Simulations. arXiv preprint arXiv:2407.10441 (2024)."},{"key":"e_1_3_2_1_15_1","unstructured":"Brendan McMahan Eider Moore Daniel Ramage Seth Hampson and Blaise Aguera y Arcas. 2017. Communication-efficient learning of deep networks from decentralized data. In Artificial intelligence and statistics. PMLR 1273--1282."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581791.3596844"},{"key":"e_1_3_2_1_17_1","volume-title":"UCF101: A Dataset of 101 Human Actions Classes From Videos in The Wild. ArXiv abs\/1212.0402","author":"Soomro Khurram","year":"2012","unstructured":"Khurram Soomro, Amir Zamir, and Mubarak Shah. 2012. UCF101: A Dataset of 101 Human Actions Classes From Videos in The Wild. ArXiv abs\/1212.0402 (2012). https:\/\/api.semanticscholar.org\/CorpusID:7197134"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2022.01.063"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3625558"},{"key":"e_1_3_2_1_20_1","volume-title":"Multimodal Federated Learning via Contrastive Representation Ensemble. ArXiv abs\/2302.08888","author":"Yu Qiying","year":"2023","unstructured":"Qiying Yu, Yang Liu, Yimu Wang, Ke Xu, and Jingjing Liu. 2023. Multimodal Federated Learning via Contrastive Representation Ensemble. ArXiv abs\/2302.08888 (2023). https:\/\/api.semanticscholar.org\/CorpusID:257019892"},{"key":"e_1_3_2_1_21_1","volume-title":"Creating a Lens of Chinese Culture: A Multimodal Dataset for Chinese Pun Rebus Art Understanding. ArXiv abs\/2406.10318","author":"Zhang Tuo","year":"2024","unstructured":"Tuo Zhang, Tiantian Feng, Yibin Ni, Mengqin Cao, Ruying Liu, Katharine Butler, Yanjun Weng, Mi Zhang, Shrikanth S. Narayanan, and Salman Avestimehr. 2024. Creating a Lens of Chinese Culture: A Multimodal Dataset for Chinese Pun Rebus Art Understanding. ArXiv abs\/2406.10318 (2024). https:\/\/api.semanticscholar.org\/CorpusID:270560580"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/IOTM.004.2100182"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/IoTDI54339.2022.00011"},{"key":"e_1_3_2_1_24_1","volume-title":"Multimodal Federated Learning on IoT Data. 2022 IEEE\/ACM Seventh International Conference on Internet-of-Things Design and Implementation (IoTDI)","author":"Zhao Yuchen","year":"2021","unstructured":"Yuchen Zhao, Payam M. Barnaghi, and Hamed Haddadi. 2021. Multimodal Federated Learning on IoT Data. 2022 IEEE\/ACM Seventh International Conference on Internet-of-Things Design and Implementation (IoTDI) (2021), 43--54. https:\/\/api.semanticscholar.org\/CorpusID:246996989"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11633-021-1293-0"}],"event":{"name":"MMSys '25: ACM Multimedia Systems Conference 2025","location":"Stellenbosch South Africa","acronym":"MMSys '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia","SIGCOMM ACM Special Interest Group on Data Communication","SIGMOBILE ACM Special Interest Group on Mobility of Systems, Users, Data and Computing"]},"container-title":["Proceedings of the 35th edition of the Workshop on Network and Operating System Support for Digital Audio and Video"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3712678.3721885","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3712678.3721885","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,26]],"date-time":"2025-08-26T19:16:27Z","timestamp":1756235787000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3712678.3721885"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,31]]},"references-count":25,"alternative-id":["10.1145\/3712678.3721885","10.1145\/3712678"],"URL":"https:\/\/doi.org\/10.1145\/3712678.3721885","relation":{},"subject":[],"published":{"date-parts":[[2025,3,31]]},"assertion":[{"value":"2025-04-03","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}