{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T18:39:29Z","timestamp":1782931169517,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":61,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Dalian Science and Technology Talent Innovation Support Plan","award":["2022RY17, 2023JJ11CG001"],"award-info":[{"award-number":["2022RY17, 2023JJ11CG001"]}]},{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U23A20386, 62276045, 62293540, 62293542"],"award-info":[{"award-number":["U23A20386, 62276045, 62293540, 62293542"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3680926","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:41Z","timestamp":1729925981000},"page":"3926-3935","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":6,"title":["SelM: Selective Mechanism based Audio-Visual Segmentation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-8907-3459","authenticated-orcid":false,"given":"Jiaxu","family":"Li","sequence":"first","affiliation":[{"name":"Dalian University of Technology, Dalian, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-6658-1385","authenticated-orcid":false,"given":"Songsong","family":"Yu","sequence":"additional","affiliation":[{"name":"Dalian University of Technology, Dalian, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1911-2526","authenticated-orcid":false,"given":"Yifan","family":"Wang","sequence":"additional","affiliation":[{"name":"Dalian University of Technology, Dalian, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2538-8358","authenticated-orcid":false,"given":"Lijun","family":"Wang","sequence":"additional","affiliation":[{"name":"Dalian University of Technology, Dalian, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6668-9758","authenticated-orcid":false,"given":"Huchuan","family":"Lu","sequence":"additional","affiliation":[{"name":"Dalian University of Technology, Dalian, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"crossref","unstructured":"Relja Arandjelovic and Andrew Zisserman. 2017. Look listen and learn. In ICCV. 609--617.","DOI":"10.1109\/ICCV.2017.73"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"crossref","unstructured":"Relja Arandjelovic and Andrew Zisserman. 2018. Objects that sound. In ECCV. 435--451.","DOI":"10.1007\/978-3-030-01246-5_27"},{"key":"e_1_3_2_1_3_1","volume-title":"On the benefits of early fusion in multimodal representation learning. arXiv preprint","author":"Barnum George","year":"2020","unstructured":"George Barnum, Sabera Talukder, and Yisong Yue. 2020. On the benefits of early fusion in multimodal representation learning. arXiv preprint (2020)."},{"key":"e_1_3_2_1_4_1","volume-title":"LOCOST: State-Space Models for Long Document Abstractive Summarization. arXiv preprint","author":"Bronnec Florian Le","year":"2024","unstructured":"Florian Le Bronnec, Song Duong, Mathieu Ravaut, Alexandre Allauzen, Nancy F Chen, Vincent Guigue, Alberto Lumbreras, Laure Soulier, and Patrick Gallinari. 2024. LOCOST: State-Space Models for Long Document Abstractive Summarization. arXiv preprint (2024)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"Honglie Chen Weidi Xie Triantafyllos Afouras Arsha Nagrani Andrea Vedaldi and Andrew Zisserman. 2021. Localizing visual sounds the hard way. In CVPR. 16867--16876.","DOI":"10.1109\/CVPR46437.2021.01659"},{"key":"e_1_3_2_1_6_1","volume-title":"Taylor","author":"Duke Brendan","year":"2021","unstructured":"Brendan Duke, Abdalla Ahmed, Christian Wolf, Parham Aarabi, and Graham W. Taylor. 2021. SSTVOS: Sparse Spatiotemporal Transformers for Video Object Segmentation. In CVPR."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.23919\/FUSION45008.2020.9190246"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i11.29104"},{"key":"e_1_3_2_1_9_1","volume-title":"Audio set: An ontology and human-labeled dataset for audio events","author":"Gemmeke Jort F","unstructured":"Jort F Gemmeke, Daniel PWEllis, Dylan Freedman, Aren Jansen,Wade Lawrence, R Channing Moore, Manoj Plakal, and Marvin Ritter. 2017. Audio set: An ontology and human-labeled dataset for audio events. In ICASSP. IEEE, 776--780."},{"key":"e_1_3_2_1_10_1","volume-title":"International Conference on Intelligent Human Computer Interaction. Springer, 689--702","author":"Ghosh Shankhanil","year":"2021","unstructured":"Shankhanil Ghosh, Chhanda Saha, Nagamani Molakathala, Souvik Ghosh, and Dhananjay Singh. 2021. reSenseNet: Ensemble early fusion deep learning architecture for multimodal sentiment analysis. In International Conference on Intelligent Human Computer Interaction. Springer, 689--702."},{"key":"e_1_3_2_1_11_1","volume-title":"Mamba: Linear-time sequence modeling with selective state spaces. arXiv preprint","author":"Gu Albert","year":"2023","unstructured":"Albert Gu and Tri Dao. 2023. Mamba: Linear-time sequence modeling with selective state spaces. arXiv preprint (2023)."},{"key":"e_1_3_2_1_12_1","first-page":"1474","article-title":"Hippo: Recurrent memory with optimal polynomial projections","volume":"33","author":"Gu Albert","year":"2020","unstructured":"Albert Gu, Tri Dao, Stefano Ermon, Atri Rudra, and Christopher R\u00e9. 2020. Hippo: Recurrent memory with optimal polynomial projections. NeurIPS 33 (2020), 1474--1487.","journal-title":"NeurIPS"},{"key":"e_1_3_2_1_13_1","first-page":"35971","article-title":"On the parameterization and initialization of diagonal state space models","volume":"35","author":"Gu Albert","year":"2022","unstructured":"Albert Gu, Karan Goel, Ankit Gupta, and Christopher R\u00e9. 2022. On the parameterization and initialization of diagonal state space models. NeurIPS 35 (2022), 35971--35983.","journal-title":"NeurIPS"},{"key":"e_1_3_2_1_14_1","volume-title":"Efficiently modeling long sequences with structured state spaces. arXiv preprint","author":"Gu Albert","year":"2021","unstructured":"Albert Gu, Karan Goel, and Christopher R\u00e9. 2021. Efficiently modeling long sequences with structured state spaces. arXiv preprint (2021)."},{"key":"e_1_3_2_1_15_1","first-page":"572","article-title":"Combining recurrent, convolutional, and continuous-time models with linear state space layers","volume":"34","author":"Gu Albert","year":"2021","unstructured":"Albert Gu, Isys Johnson, Karan Goel, Khaled Saab, Tri Dao, Atri Rudra, and Christopher R\u00e9. 2021. Combining recurrent, convolutional, and continuous-time models with linear state space layers. NeurIPS 34 (2021), 572--585.","journal-title":"NeurIPS"},{"key":"e_1_3_2_1_16_1","volume-title":"MambaIR: A Simple Baseline for Image Restoration with State-Space Model. arXiv preprint","author":"Guo Hang","year":"2024","unstructured":"Hang Guo, Jinmin Li, Tao Dai, Zhihao Ouyang, Xudong Ren, and Shu-Tao Xia. 2024. MambaIR: A Simple Baseline for Image Restoration with State-Space Model. arXiv preprint (2024)."},{"key":"e_1_3_2_1_17_1","first-page":"22982","article-title":"Diagonal state spaces are as effective as structured state spaces","volume":"35","author":"Gupta Ankit","year":"2022","unstructured":"Ankit Gupta, Albert Gu, and Jonathan Berant. 2022. Diagonal state spaces are as effective as structured state spaces. NeurIPS 35 (2022), 22982--22994.","journal-title":"NeurIPS"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i3.27978"},{"key":"e_1_3_2_1_19_1","volume-title":"Liquid structural state-space models. arXiv preprint","author":"Hasani Ramin","year":"2022","unstructured":"Ramin Hasani, Mathias Lechner, Tsun-HsuanWang, Makram Chahine, Alexander Amini, and Daniela Rus. 2022. Liquid structural state-space models. arXiv preprint (2022)."},{"key":"e_1_3_2_1_20_1","unstructured":"Junwen He Yifan Wang Lijun Wang Huchuan Lu Bin Luo Jun-Yan He Jin-Peng Lan Yifeng Geng and Xuansong Xie. 2023. Towards Deeply Unified Depth-aware Panoptic Segmentation with Bi-directional Guidance Learning. In ICCV. 4111--4121."},{"key":"e_1_3_2_1_21_1","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep residual learning for image recognition. In CVPR. 770--778."},{"key":"e_1_3_2_1_22_1","volume-title":"Gaussian error linear units (gelus). arXiv preprint","author":"Hendrycks Dan","year":"2016","unstructured":"Dan Hendrycks and Kevin Gimpel. 2016. Gaussian error linear units (gelus). arXiv preprint (2016)."},{"key":"e_1_3_2_1_23_1","volume-title":"Jort F Gemmeke, Aren Jansen, R Channing Moore, Manoj Plakal, Devin Platt, Rif A Saurous, Bryan Seybold, et al.","author":"Hershey Shawn","year":"2017","unstructured":"Shawn Hershey, Sourish Chaudhuri, Daniel PW Ellis, Jort F Gemmeke, Aren Jansen, R Channing Moore, Manoj Plakal, Devin Platt, Rif A Saurous, Bryan Seybold, et al. 2017. CNN architectures for large-scale audio classification. In ICASSP. IEEE, 131--135."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"crossref","unstructured":"Di Hu Feiping Nie and Xuelong Li. 2019. Deep multimodal clustering for unsupervised audiovisual learning. In CVPR. 9248--9257.","DOI":"10.1109\/CVPR.2019.00947"},{"key":"e_1_3_2_1_25_1","first-page":"10077","article-title":"Discriminative sounding objects localization via self-supervised audiovisual matching","volume":"33","author":"Hu Di","year":"2020","unstructured":"Di Hu, Rui Qian, Minyue Jiang, Xiao Tan, Shilei Wen, Errui Ding, Weiyao Lin, and Dejing Dou. 2020. Discriminative sounding objects localization via self-supervised audiovisual matching. NeurIPS 33 (2020), 10077--10087.","journal-title":"NeurIPS"},{"key":"e_1_3_2_1_26_1","volume-title":"Ming Gui, Olga Grebenkova, Pingchuan Ma, Johannes Fischer, and Bjorn Ommer.","author":"Hu Vincent Tao","year":"2024","unstructured":"Vincent Tao Hu, Stefan Andreas Baumann, Ming Gui, Olga Grebenkova, Pingchuan Ma, Johannes Fischer, and Bjorn Ommer. 2024. ZigMa: Zigzag Mamba Diffusion Model. arXiv preprint (2024)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","unstructured":"Rudolph Emil Kalman. 1960. A new approach to linear filtering and prediction problems.(1960).","DOI":"10.1115\/1.3662552"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"crossref","unstructured":"Alexander Kirillov Ross Girshick Kaiming He and Piotr Doll\u00e1r. 2019. Panoptic feature pyramid networks. In CVPR. 6399--6408.","DOI":"10.1109\/CVPR.2019.00656"},{"key":"e_1_3_2_1_29_1","volume-title":"Catr: Combinatorial-dependence audio-queried transformer for audio-visual video segmentation. In ACM MM. 1485--1494.","author":"Li Kexin","year":"2023","unstructured":"Kexin Li, Zongxin Yang, Lei Chen, Yi Yang, and Jun Xiao. 2023. Catr: Combinatorial-dependence audio-queried transformer for audio-visual video segmentation. In ACM MM. 1485--1494."},{"key":"e_1_3_2_1_30_1","first-page":"5801","article-title":"From pixels to semantics: self-supervised video object segmentation with multiperspective feature mining","volume":"31","author":"Li Ruoqi","year":"2022","unstructured":"Ruoqi Li, Yifan Wang, Lijun Wang, Huchuan Lu, Xiaopeng Wei, and Qiang Zhang. 2022. From pixels to semantics: self-supervised video object segmentation with multiperspective feature mining. IEEE TIP 31 (2022), 5801--5812.","journal-title":"IEEE TIP"},{"key":"e_1_3_2_1_31_1","volume-title":"PointMamba: A Simple State Space Model for Point Cloud Analysis. arXiv preprint","author":"Liang Dingkang","year":"2024","unstructured":"Dingkang Liang, Xin Zhou, Xinyu Wang, Xingkui Zhu, Wei Xu, Zhikang Zou, Xiaoqing Ye, and Xiang Bai. 2024. PointMamba: A Simple State Space Model for Point Cloud Analysis. arXiv preprint (2024)."},{"key":"e_1_3_2_1_32_1","volume-title":"Xingqun Qi, Hu Zhang, Lincheng Li, Dadong Wang, and Xin Yu.","author":"Liu Chen","year":"2023","unstructured":"Chen Liu, Peike Patrick Li, Xingqun Qi, Hu Zhang, Lincheng Li, Dadong Wang, and Xin Yu. 2023. Audio-Visual Segmentation by Exploring Cross-Modal Mutual Semantics. In ACM MM. 7590--7598."},{"key":"e_1_3_2_1_33_1","volume-title":"Audio-aware query-enhanced transformer for audio-visual segmentation. arXiv preprint","author":"Liu Jinxiang","year":"2023","unstructured":"Jinxiang Liu, Chen Ju, Chaofan Ma, Yanfeng Wang, Yu Wang, and Ya Zhang. 2023. Audio-aware query-enhanced transformer for audio-visual segmentation. arXiv preprint (2023)."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i2.20073"},{"key":"e_1_3_2_1_35_1","volume-title":"Vmamba: Visual state space model. arXiv preprint","author":"Liu Yue","year":"2024","unstructured":"Yue Liu, Yunjie Tian, Yuzhong Zhao, Hongtian Yu, Lingxi Xie, Yaowei Wang, Qixiang Ye, and Yunfan Liu. 2024. Vmamba: Visual state space model. arXiv preprint (2024)."},{"key":"e_1_3_2_1_36_1","volume-title":"Mega: moving average equipped gated attention. arXiv preprint","author":"Ma Xuezhe","year":"2022","unstructured":"Xuezhe Ma, Chunting Zhou, Xiang Kong, Junxian He, Liangke Gui, Graham Neubig, Jonathan May, and Luke Zettlemoyer. 2022. Mega: moving average equipped gated attention. arXiv preprint (2022)."},{"key":"e_1_3_2_1_37_1","unstructured":"Sabarinath Mahadevan Ali Athar Sebastian Hennen Laura Leal-Taix\u00e9 and Bastian Leibe. 2023. Making a Case for 3D Convolutions for Object Segmentation in Videos. arXiv:2008.11516 [cs.CV]"},{"key":"e_1_3_2_1_38_1","volume-title":"Transformer transforms salient object detection and camouflaged object detection. arXiv preprint arXiv:2104.10127","author":"Mao Yuxin","year":"2021","unstructured":"Yuxin Mao, Jing Zhang, Zhexiong Wan, Yuchao Dai, Aixuan Li, Yunqiu Lv, Xinyu Tian, Deng-Ping Fan, and Nick Barnes. 2021. Transformer transforms salient object detection and camouflaged object detection. arXiv preprint arXiv:2104.10127 (2021)."},{"key":"e_1_3_2_1_39_1","volume-title":"Contrastive conditional latent diffusion for audio-visual segmentation. arXiv preprint","author":"Mao Yuxin","year":"2023","unstructured":"Yuxin Mao, Jing Zhang, Mochu Xiang, Yunqiu Lv, Yiran Zhong, and Yuchao Dai. 2023. Contrastive conditional latent diffusion for audio-visual segmentation. arXiv preprint (2023)."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"crossref","unstructured":"Yuxin Mao Jing Zhang Mochu Xiang Yiran Zhong and Yuchao Dai. 2023. Multimodal variational auto-encoder based audio-visual segmentation. In ICCV. 954--965.","DOI":"10.1109\/ICCV51070.2023.00094"},{"key":"e_1_3_2_1_41_1","volume-title":"Long range language modeling via gated state spaces. arXiv preprint","author":"Mehta Harsh","year":"2022","unstructured":"Harsh Mehta, Ankit Gupta, Ashok Cutkosky, and Behnam Neyshabur. 2022. Long range language modeling via gated state spaces. arXiv preprint (2022)."},{"key":"e_1_3_2_1_42_1","volume-title":"V-net: Fully convolutional neural networks for volumetric medical image segmentation. In 2016 fourth international conference on 3D vision (3DV). Ieee, 565--571.","author":"Milletari Fausto","year":"2016","unstructured":"Fausto Milletari, Nassir Navab, and Seyed-Ahmad Ahmadi. 2016. V-net: Fully convolutional neural networks for volumetric medical image segmentation. In 2016 fourth international conference on 3D vision (3DV). Ieee, 565--571."},{"key":"e_1_3_2_1_43_1","volume-title":"Diagnosing Alzheimer's Disease using Early-Late Multimodal Data Fusion with Jacobian Maps. arXiv preprint","author":"Mustafa Yasmine","year":"2023","unstructured":"Yasmine Mustafa and Tie Luo. 2023. Diagnosing Alzheimer's Disease using Early-Late Multimodal Data Fusion with Jacobian Maps. arXiv preprint (2023)."},{"key":"e_1_3_2_1_44_1","volume-title":"Efficientvmamba: Atrous selective scan for light weight visual mamba. arXiv preprint","author":"Pei Xiaohuan","year":"2024","unstructured":"Xiaohuan Pei, Tao Huang, and Chang Xu. 2024. Efficientvmamba: Atrous selective scan for light weight visual mamba. arXiv preprint (2024)."},{"key":"e_1_3_2_1_45_1","volume-title":"Multiple sound sources localization from coarse to fine","author":"Qian Rui","unstructured":"Rui Qian, Di Hu, Heinrich Dinkel, Mengyue Wu, Ning Xu, and Weiyao Lin. 2020. Multiple sound sources localization from coarse to fine. In ECCV. Springer, 292--308."},{"key":"e_1_3_2_1_46_1","volume-title":"Self-supervised audio-visual co-segmentation","author":"Rouditchenko Andrew","unstructured":"Andrew Rouditchenko, Hang Zhao, Chuang Gan, Josh McDermott, and Antonio Torralba. 2019. Self-supervised audio-visual co-segmentation. In ICASSP. IEEE, 2357--2361."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"crossref","unstructured":"Arda Senocak Tae-Hyun Oh Junsik Kim Ming-Hsuan Yang and In So Kweon. 2018. Learning to localize sound source in visual scenes. In CVPR. 4358--4366.","DOI":"10.1109\/CVPR.2018.00458"},{"key":"e_1_3_2_1_49_1","volume-title":"Simplified state space layers for sequence modeling. arXiv preprint","author":"Smith Jimmy TH","year":"2022","unstructured":"Jimmy TH Smith, Andrew Warrington, and Scott W Linderman. 2022. Simplified state space layers for sequence modeling. arXiv preprint (2022)."},{"key":"e_1_3_2_1_50_1","volume-title":"Dropout: a simple way to prevent neural networks from overfitting. The journal of machine learning research 15, 1","author":"Srivastava Nitish","year":"2014","unstructured":"Nitish Srivastava, Geoffrey Hinton, Alex Krizhevsky, Ilya Sutskever, and Ruslan Salakhutdinov. 2014. Dropout: a simple way to prevent neural networks from overfitting. The journal of machine learning research 15, 1 (2014), 1929--1958."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475587"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1007\/s41095-022-0274-8"},{"key":"e_1_3_2_1_53_1","volume-title":"Cross-Modal Relation-Aware Networks for Audio-Visual Event Localization. In ACM International Conference on Multimedia.","author":"Xu Haoming","year":"2020","unstructured":"Haoming Xu, Runhao Zeng, Qingyao Wu, Mingkui Tan, and Chuang Gan. 2020. Cross-Modal Relation-Aware Networks for Audio-Visual Event Localization. In ACM International Conference on Multimedia."},{"key":"e_1_3_2_1_54_1","unstructured":"Zongxin Yang Yunchao Wei and Yi Yang. 2021. Associating Objects with Transformers for Video Object Segmentation. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547869"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-981-16-8531-6_8"},{"key":"e_1_3_2_1_57_1","volume-title":"Learning Generative Vision Transformer with Energy-Based Latent Space for Saliency Prediction. In 2021 Conference on Neural Information Processing Systems.","author":"Zhang Jing","year":"2021","unstructured":"Jing Zhang, Jianwen Xie, Nick Barnes, and Ping Li. 2021. Learning Generative Vision Transformer with Energy-Based Latent Space for Saliency Prediction. In 2021 Conference on Neural Information Processing Systems."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"crossref","unstructured":"Haojie Zhao Junsong Chen Lijun Wang and Huchuan Lu. 2023. Arkittrack: a new diverse dataset for tracking using mobile RGB-D data. In CVPR. 5126--5135.","DOI":"10.1109\/CVPR52729.2023.00496"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"crossref","unstructured":"Jinxing Zhou Xuyang Shen Jianyuan Wang Jiayi Zhang Weixuan Sun Jing Zhang Stan Birchfield Dan Guo Lingpeng Kong Meng Wang et al. 2023. Audio-visual segmentation with semantics. arXiv preprint (2023).","DOI":"10.1007\/s11263-024-02261-x"},{"key":"e_1_3_2_1_60_1","volume-title":"Audio-visual segmentation","author":"Zhou Jinxing","unstructured":"Jinxing Zhou, Jianyuan Wang, Jiayi Zhang, Weixuan Sun, Jing Zhang, Stan Birchfield, Dan Guo, Lingpeng Kong, Meng Wang, and Yiran Zhong. 2022. Audio-visual segmentation. In ECCV. Springer, 386--403."},{"key":"e_1_3_2_1_61_1","volume-title":"Vision mamba: Efficient visual representation learning with bidirectional state space model. arXiv preprint","author":"Zhu Lianghui","year":"2024","unstructured":"Lianghui Zhu, Bencheng Liao, Qian Zhang, Xinlong Wang, Wenyu Liu, and Xinggang Wang. 2024. Vision mamba: Efficient visual representation learning with bidirectional state space model. arXiv preprint (2024)."}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3680926","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3680926","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:34Z","timestamp":1750295854000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3680926"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":61,"alternative-id":["10.1145\/3664647.3680926","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3680926","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}