{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,22]],"date-time":"2025-10-22T00:53:05Z","timestamp":1761094385727,"version":"build-2065373602"},"publisher-location":"New York, NY, USA","reference-count":36,"publisher":"ACM","funder":[{"name":"National Key R&amp;D Program of China","award":["2022ZD0119100"],"award-info":[{"award-number":["2022ZD0119100"]}]},{"name":"National Natural Science Foundation of China","award":["62476238, 62202436"],"award-info":[{"award-number":["62476238, 62202436"]}]},{"name":"Natural Science Foundation of Zhejiang Province","award":["LY24F020012, LHZSD24F020001"],"award-info":[{"award-number":["LY24F020012, LHZSD24F020001"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746266.3762161","type":"proceedings-article","created":{"date-parts":[[2025,10,21]],"date-time":"2025-10-21T15:19:13Z","timestamp":1761059953000},"page":"21-29","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Mamba-Based Multimodal Continual Learning for Audio-Visual Classification with Prototype-Enhanced Anti-Forgetting Mechanism"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-0100-018X","authenticated-orcid":false,"given":"Jingyang","family":"Lin","sequence":"first","affiliation":[{"name":"School of Computer and Computing Science, Hangzhou City University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-4831-9220","authenticated-orcid":false,"given":"Xinru","family":"Ying","sequence":"additional","affiliation":[{"name":"School of Computer and Computing Science, Hangzhou City University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-2728-1751","authenticated-orcid":false,"given":"Jiaqi","family":"Mo","sequence":"additional","affiliation":[{"name":"College of Letters &amp; Science, University of Wisconsin\u2013Madison, Madison, WI, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-0077-2905","authenticated-orcid":false,"given":"Lina","family":"Wei","sequence":"additional","affiliation":[{"name":"Hangzhou City University, Hangzhou, China and Zhejiang Provincial Engineering Research Center for Real-Time SmartTech in Urban Security Governance, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7685-5705","authenticated-orcid":false,"given":"Fangfang","family":"Wang","sequence":"additional","affiliation":[{"name":"School of Information Science and Technology, Hangzhou Normal University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9774-9688","authenticated-orcid":false,"given":"Canghong","family":"Jin","sequence":"additional","affiliation":[{"name":"Hangzhou City University, Hangzhou, China and Zhejiang Provincial Engineering Research Center for Real-Time SmartTech in Urban Security Governance, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4255-658X","authenticated-orcid":false,"given":"Guanlin","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Computer and Computing Science, Hangzhou City University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,26]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.1611835114"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00458"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01216-8_16"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01805"},{"key":"e_1_3_2_1_5_1","first-page":"10553","volume-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","author":"Mercea Otniel-Bogdan","year":"2022","unstructured":"Otniel-Bogdan Mercea, Lukas Riesch, A Koepke, and Zeynep Akata. Audio-visual generalised zero-shot learning with cross-modal attention and language. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pages 10553-10563, 2022."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.73"},{"key":"e_1_3_2_1_7_1","volume-title":"Vatt: Transformers for multimodal self-supervised learning from raw video, audio and text. Advances in neural information processing systems, 34:24206-24221","author":"Akbari Hassan","year":"2021","unstructured":"Hassan Akbari, Liangzhe Yuan, Rui Qian, Wei-Hong Chuang, Shih-Fu Chang, Yin Cui, and Boqing Gong. Vatt: Transformers for multimodal self-supervised learning from raw video, audio and text. Advances in neural information processing systems, 34:24206-24221, 2021."},{"key":"e_1_3_2_1_8_1","volume-title":"Attention is all you need. Advances in neural information processing systems, 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. Attention is all you need. Advances in neural information processing systems, 30, 2017."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00639"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01936"},{"key":"e_1_3_2_1_11_1","first-page":"127226","article-title":"Theoretical foundations of deep selective state-space models","volume":"37","author":"Cirone Nicola Muca","year":"2024","unstructured":"Nicola Muca Cirone, Antonio Orvieto, Benjamin Walker, Cristopher Salvi, and Terry Lyons. Theoretical foundations of deep selective state-space models. Advances in Neural Information Processing Systems, 37:127226-127272, 2024.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_12_1","volume-title":"Mamba: Linear-time sequence modeling with selective state spaces. arXiv preprint arXiv:2312.00752","author":"Gu Albert","year":"2023","unstructured":"Albert Gu and Tri Dao. Mamba: Linear-time sequence modeling with selective state spaces. arXiv preprint arXiv:2312.00752, 2023."},{"key":"e_1_3_2_1_13_1","volume-title":"Learning without forgetting","author":"Li Zhizhong","year":"2017","unstructured":"Zhizhong Li and Derek Hoiem. Learning without forgetting. IEEE transactions on pattern analysis and machine intelligence, 40(12):2935-2947, 2017."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.587"},{"key":"e_1_3_2_1_15_1","volume-title":"Hippo: Recurrent memory with optimal polynomial projections. Advances in neural information processing systems, 33:1474-1487","author":"Gu Albert","year":"2020","unstructured":"Albert Gu, Tri Dao, Stefano Ermon, Atri Rudra, and Christopher R\u00e9. Hippo: Recurrent memory with optimal polynomial projections. Advances in neural information processing systems, 33:1474-1487, 2020."},{"key":"e_1_3_2_1_16_1","volume-title":"Diagonal state spaces are as effective as structured state spaces. Advances in neural information processing systems, 35:22982-22994","author":"Gupta Ankit","year":"2022","unstructured":"Ankit Gupta, Albert Gu, and Jonathan Berant. Diagonal state spaces are as effective as structured state spaces. Advances in neural information processing systems, 35:22982-22994, 2022."},{"key":"e_1_3_2_1_17_1","first-page":"237","volume-title":"European conference on computer vision","author":"Li Kunchang","year":"2024","unstructured":"Kunchang Li, Xinhao Li, Yi Wang, Yinan He, Yali Wang, Limin Wang, and Yu Qiao. Videomamba: State space model for efficient video understanding. In European conference on computer vision, pages 237-255. Springer, 2024."},{"key":"e_1_3_2_1_18_1","article-title":"Exploring temporal and multi-modal mamba for audio-visual segmentation","author":"Gong Sitong","year":"2025","unstructured":"Sitong Gong, Yunzhi Zhuge, Lu Zhang, Yifan Wang, Pingping Zhang, Lijun Wang, and Huchuan Lu. Avs-mamba: Exploring temporal and multi-modal mamba for audio-visual segmentation. IEEE Transactions on Multimedia, 2025.","journal-title":"IEEE Transactions on Multimedia"},{"key":"e_1_3_2_1_19_1","volume-title":"Prototypical networks for few-shot learning. Advances in neural information processing systems, 30","author":"Snell Jake","year":"2017","unstructured":"Jake Snell, Kevin Swersky, and Richard Zemel. Prototypical networks for few-shot learning. Advances in neural information processing systems, 30, 2017."},{"key":"e_1_3_2_1_20_1","first-page":"11674","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"Malepathirana Tamasha","year":"2023","unstructured":"Tamasha Malepathirana, Damith Senanayake, and Saman Halgamuge. Napa-vq: Neighborhood-aware prototype augmentation with vector quantization for continual learning. In Proceedings of the IEEE\/CVF International Conference on Computer Vision, pages 11674-11684, 2023."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00170"},{"key":"e_1_3_2_1_22_1","first-page":"15060","article-title":"Few-shot class-incremental learning via training-free prototype calibration","volume":"36","author":"Wang Qi-Wei","year":"2023","unstructured":"Qi-Wei Wang, Da-Wei Zhou, Yi-Kai Zhang, De-Chuan Zhan, and Han-Jia Ye. Few-shot class-incremental learning via training-free prototype calibration. Advances in Neural Information Processing Systems, 36:15060-15076, 2023.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01457"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01246-5_35"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01246-5_27"},{"key":"e_1_3_2_1_26_1","volume-title":"Contrastive audio-visual masked autoencoder. arXiv preprint arXiv:2210.07839","author":"Gong Yuan","year":"2022","unstructured":"Yuan Gong, Andrew Rouditchenko, Alexander H Liu, David Harwath, Leonid Karlinsky, Hilde Kuehne, and James Glass. Contrastive audio-visual masked autoencoder. arXiv preprint arXiv:2210.07839, 2022."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2024.110273"},{"key":"e_1_3_2_1_28_1","volume-title":"Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531","author":"Hinton Geoffrey","year":"2015","unstructured":"Geoffrey Hinton, Oriol Vinyals, and Jeff Dean. Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531, 2015."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58565-5_6"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00717"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02191"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3213473"},{"key":"e_1_3_2_1_33_1","volume-title":"Vision mamba: Efficient visual representation learning with bidirectional state space model. arXiv preprint arXiv:2401.09417","author":"Zhu Lianghui","year":"2024","unstructured":"Lianghui Zhu, Bencheng Liao, Qian Zhang, Xinlong Wang, Wenyu Liu, and Xinggang Wang. Vision mamba: Efficient visual representation learning with bidirectional state space model. arXiv preprint arXiv:2401.09417, 2024."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053174"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00088"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01560"}],"event":{"name":"MM '25:The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 3rd International Workshop on Deep Multimodal Generation and Retrieval"],"original-title":[],"deposited":{"date-parts":[[2025,10,21]],"date-time":"2025-10-21T15:20:14Z","timestamp":1761060014000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746266.3762161"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,26]]},"references-count":36,"alternative-id":["10.1145\/3746266.3762161","10.1145\/3746266"],"URL":"https:\/\/doi.org\/10.1145\/3746266.3762161","relation":{},"subject":[],"published":{"date-parts":[[2025,10,26]]},"assertion":[{"value":"2025-10-26","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}