{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,7]],"date-time":"2026-05-07T15:12:52Z","timestamp":1778166772074,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":94,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100018537","name":"National Science and Technology Major Project","doi-asserted-by":"publisher","award":["2022ZD0118201"],"award-info":[{"award-number":["2022ZD0118201"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100018537","id-type":"DOI","asserted-by":"publisher"}]},{"name":"the National Science Fund for Distinguished Young Scholars","award":["62025603"],"award-info":[{"award-number":["62025603"]}]},{"name":"the National Natural Science Foundation of China","award":["U21B2037, U22B2051, 623B2088, 62176222, 62176223, 62176226, 62072386, 62072387, 62072389, 62002305 and 62272401"],"award-info":[{"award-number":["U21B2037, U22B2051, 623B2088, 62176222, 62176223, 62176226, 62072386, 62072387, 62072389, 62002305 and 62272401"]}]},{"name":"the Natural Science Foundation of Fujian Province of China","award":["2021J01002, and 2022J06001"],"award-info":[{"award-number":["2021J01002, and 2022J06001"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3680571","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:41Z","timestamp":1729925981000},"page":"905-914","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Deep Instruction Tuning for Segment Anything Model"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-3168-1491","authenticated-orcid":false,"given":"Xiaorui","family":"Huang","sequence":"first","affiliation":[{"name":"Key Laboratory of Multimedia Trusted Perception and Efficient Computing, Ministry of Education of China, Xiamen University, Xiamen, Fujian, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5334-1843","authenticated-orcid":false,"given":"Gen","family":"Luo","sequence":"additional","affiliation":[{"name":"Key Laboratory of Multimedia Trusted Perception and Efficient Computing, Ministry of Education of China, Xiamen University, Xiamen, Fujian, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-7455-326X","authenticated-orcid":false,"given":"Chaoyang","family":"Zhu","sequence":"additional","affiliation":[{"name":"The Department of Computer Science and Engineering, The Hong Kong University of Science and Technology, Hong Kong, Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-3569-4278","authenticated-orcid":false,"given":"Bo","family":"Tong","sequence":"additional","affiliation":[{"name":"Key Laboratory of Multimedia Trusted Perception and Efficient Computing, Minis, Xiamen University, Xiamen, Fujian, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5110-4526","authenticated-orcid":false,"given":"Yiyi","family":"Zhou","sequence":"additional","affiliation":[{"name":"Key Laboratory of Multimedia Trusted Perception and Efficient Computing, Minis, Xiamen University, Xiamen, Fujian, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3912-9306","authenticated-orcid":false,"given":"Xiaoshuai","family":"Sun","sequence":"additional","affiliation":[{"name":"Key Laboratory of Multimedia Trusted Perception and Efficient Computing, Minis, Xiamen University, Xiamen, Fujian, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9163-2932","authenticated-orcid":false,"given":"Rongrong","family":"Ji","sequence":"additional","affiliation":[{"name":"Key Laboratory of Multimedia Trusted Perception and Efficient Computing, Ministry of Education of China, Xiamen University, Xiamen, Fujian, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Segnet: A deep convolutional encoder-decoder architecture for image segmentation. TPAMI","author":"Badrinarayanan Vijay","year":"2017","unstructured":"Vijay Badrinarayanan, Alex Kendall, and Roberto Cipolla. 2017. Segnet: A deep convolutional encoder-decoder architecture for image segmentation. TPAMI (2017)."},{"key":"e_1_3_2_1_2_1","volume-title":"Deeplab: Semantic image segmentation with deep convolutional nets, atrous convolution, and fully connected crfs. TPAMI","author":"Chen Liang-Chieh","year":"2017","unstructured":"Liang-Chieh Chen, George Papandreou, Iasonas Kokkinos, Kevin Murphy, and Alan L Yuille. 2017. Deeplab: Semantic image segmentation with deep convolutional nets, atrous convolution, and fully connected crfs. TPAMI (2017)."},{"key":"e_1_3_2_1_3_1","volume-title":"Sam-adapter: Adapting segment anything in underperformed scenes. In ICCV.","author":"Chen Tianrun","year":"2023","unstructured":"Tianrun Chen, Lanyun Zhu, Chaotao Deng, Runlong Cao, Yan Wang, Shangzhan Zhang, Zejian Li, Lingyun Sun, Ying Zang, and Papa Mao. 2023. Sam-adapter: Adapting segment anything in underperformed scenes. In ICCV."},{"key":"e_1_3_2_1_4_1","volume-title":"Brian Price, Alexander Schwing, and Joon-Young Lee.","author":"Cheng Ho Kei","year":"2023","unstructured":"Ho Kei Cheng, Seoung Wug Oh, Brian Price, Alexander Schwing, and Joon-Young Lee. 2023. Tracking anything with decoupled video segmentation. In ICCV. 1316--1326."},{"key":"e_1_3_2_1_5_1","volume-title":"Segment and track anything. arXiv","author":"Cheng Yangming","year":"2023","unstructured":"Yangming Cheng, Liulei Li, Yuanyou Xu, Xiaodi Li, Zongxin Yang, Wenguan Wang, and Yi Yang. 2023. Segment and track anything. arXiv (2023)."},{"key":"e_1_3_2_1_6_1","volume-title":"Unleashing the Potential of SAM for Medical Adaptation via Hierarchical Decoding. arXiv","author":"Cheng Zhiheng","year":"2024","unstructured":"Zhiheng Cheng, Qingyue Wei, Hongru Zhu, Yan Wang, Liangqiong Qu, Wei Shao, and Yuyin Zhou. 2024. Unleashing the Potential of SAM for Medical Adaptation via Hierarchical Decoding. arXiv (2024)."},{"key":"e_1_3_2_1_7_1","volume-title":"Transvg: End-to-end visual grounding with transformers. In ICCV.","author":"Deng Jiajun","year":"2021","unstructured":"Jiajun Deng, Zhengyuan Yang, Tianlang Chen, Wengang Zhou, and Houqiang Li. 2021. Transvg: End-to-end visual grounding with transformers. In ICCV."},{"key":"e_1_3_2_1_8_1","volume-title":"Xiaowei Hu, and Jing Qin.","author":"Deng Sen","year":"2024","unstructured":"Sen Deng, Yidan Feng, Haoneng Lin, Yiting Fan, Alex Pui-Wai Lee, Xiaowei Hu, and Jing Qin. 2024. Semi-supervised TEE Segmentation via Interacting with SAM Equipped with Noise-Resilient Prompting. In AAAI."},{"key":"e_1_3_2_1_9_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv (2018)."},{"key":"e_1_3_2_1_10_1","volume-title":"Vlt: Vision-language transformer and query generation for referring segmentation. TPAMI","author":"Ding Henghui","year":"2022","unstructured":"Henghui Ding, Chang Liu, Suchen Wang, and Xudong Jiang. 2022. Vlt: Vision-language transformer and query generation for referring segmentation. TPAMI (2022)."},{"key":"e_1_3_2_1_11_1","unstructured":"Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly et al. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv (2020)."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"crossref","unstructured":"Guang Feng Zhiwei Hu Lihe Zhang and Huchuan Lu. 2021. Encoder fusion network with co-attention embedding for referring image segmentation. In CVPR.","DOI":"10.1109\/CVPR46437.2021.01525"},{"key":"e_1_3_2_1_13_1","volume-title":"Editanything: Empowering unparalleled flexibility in image editing and generation. In ACM MM.","author":"Gao Shanghua","year":"2023","unstructured":"Shanghua Gao, Zhijie Lin, Xingyu Xie, Pan Zhou, Ming-Ming Cheng, and Shuicheng Yan. 2023. Editanything: Empowering unparalleled flexibility in image editing and generation. In ACM MM."},{"key":"e_1_3_2_1_14_1","volume-title":"Ppt: Pre-trained prompt tuning for few-shot learning. arXiv","author":"Gu Yuxian","year":"2021","unstructured":"Yuxian Gu, Xu Han, Zhiyuan Liu, and Minlie Huang. 2021. Ppt: Pre-trained prompt tuning for few-shot learning. arXiv (2021)."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"crossref","unstructured":"Aimee Guo Grace Fei Hemanth Pasupuleti and Jing Wang. 2024. ClickSAM: Fine-tuning Segment Anything Model using click prompts for ultrasound image segmentation. In UIT.","DOI":"10.1117\/12.3005879"},{"key":"e_1_3_2_1_16_1","volume-title":"Ptr: Prompt tuning with rules for text classification. AI Open","author":"Han Xu","year":"2022","unstructured":"Xu Han, Weilin Zhao, Ning Ding, Zhiyuan Liu, and Maosong Sun. 2022. Ptr: Prompt tuning with rules for text classification. AI Open (2022)."},{"key":"e_1_3_2_1_17_1","unstructured":"Kaiming He Georgia Gkioxari Piotr Doll\u00e1r and Ross Girshick. 2017. Mask r-cnn. In ICCV."},{"key":"e_1_3_2_1_18_1","unstructured":"Yutao Hu Qixiong Wang Wenqi Shao Enze Xie Zhenguo Li Jungong Han and Ping Luo. 2023. Beyond One-to-One: Rethinking the Referring Image Segmentation. In ICCV."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"crossref","unstructured":"Shaofei Huang Tianrui Hui Si Liu Guanbin Li Yunchao Wei Jizhong Han Luoqi Liu and Bo Li. 2020. Referring image segmentation via cross-modal progressive comprehension. In CVPR.","DOI":"10.1109\/CVPR42600.2020.01050"},{"key":"e_1_3_2_1_20_1","unstructured":"Menglin Jia Luming Tang Bor-Chun Chen Claire Cardie Serge Belongie Bharath Hariharan and Ser-Nam Lim. 2022. Visual prompt tuning. In ECCV."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"crossref","unstructured":"Yang Jiao Zequn Jie Weixin Luo Jingjing Chen Yu-Gang Jiang Xiaolin Wei and Lin Ma. 2021. Two-stage visual cues enhancement network for referring image segmentation. In ACM MM.","DOI":"10.1145\/3474085.3475222"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"crossref","unstructured":"Ya Jing Tao Kong Wei Wang Liang Wang Lei Li and Tieniu Tan. 2021. Locate then segment: A strong pipeline for referring image segmentation. In CVPR.","DOI":"10.1109\/CVPR46437.2021.00973"},{"key":"e_1_3_2_1_23_1","volume-title":"Referitgame: Referring to objects in photographs of natural scenes. In EMNLP.","author":"Kazemzadeh Sahar","year":"2014","unstructured":"Sahar Kazemzadeh, Vicente Ordonez, Mark Matten, and Tamara Berg. 2014. Referitgame: Referring to objects in photographs of natural scenes. In EMNLP."},{"key":"e_1_3_2_1_24_1","volume-title":"Restr: Convolution-free referring image segmentation using transformers. In CVPR.","author":"Kim Namyup","year":"2022","unstructured":"Namyup Kim, Dongwon Kim, Cuiling Lan, Wenjun Zeng, and Suha Kwak. 2022. Restr: Convolution-free referring image segmentation using transformers. In CVPR."},{"key":"e_1_3_2_1_25_1","volume-title":"Vilt: Vision-and-language transformer without convolution or region supervision. In ICML.","author":"Kim Wonjae","year":"2021","unstructured":"Wonjae Kim, Bokyung Son, and Ildoo Kim. 2021. Vilt: Vision-and-language transformer without convolution or region supervision. In ICML."},{"key":"e_1_3_2_1_26_1","volume-title":"Adam: A method for stochastic optimization. arXiv","author":"Kingma Diederik P","year":"2014","unstructured":"Diederik P Kingma and Jimmy Ba. 2014. Adam: A method for stochastic optimization. arXiv (2014)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","unstructured":"Alexander Kirillov Kaiming He Ross Girshick Carsten Rother and Piotr Doll\u00e1r. 2019. Panoptic segmentation. In CVPR.","DOI":"10.1109\/CVPR.2019.00963"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"crossref","unstructured":"Alexander Kirillov Eric Mintun Nikhila Ravi Hanzi Mao Chloe Rolland Laura Gustafson Tete Xiao Spencer Whitehead Alexander C Berg Wan-Yen Lo et al. 2023. Segment anything. arXiv (2023).","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"crossref","unstructured":"Ranjay Krishna Yuke Zhu Oliver Groth Justin Johnson Kenji Hata Joshua Kravitz Stephanie Chen Yannis Kalantidis Li-Jia Li David A Shamma et al. 2017. Visual genome: Connecting language and vision using crowdsourced dense image annotations. IJCV (2017).","DOI":"10.1007\/s11263-016-0981-7"},{"key":"e_1_3_2_1_30_1","volume-title":"Lisa: Reasoning segmentation via large language model. arXiv","author":"Lai Xin","year":"2023","unstructured":"Xin Lai, Zhuotao Tian, Yukang Chen, Yanwei Li, Yuhui Yuan, Shu Liu, and Jiaya Jia. 2023. Lisa: Reasoning segmentation via large language model. arXiv (2023)."},{"key":"e_1_3_2_1_31_1","volume":"202","author":"Lee Dongjun","unstructured":"Dongjun Lee, Seokwon Song, Jihee Suh, Joonmyeong Choi, Sanghyeok Lee, and Hyunwoo J Kim. 2023. Read-only Prompt Optimization for Vision-Language Few-shot Learning. In ICCV.","journal-title":"Hyunwoo J Kim."},{"key":"e_1_3_2_1_32_1","volume-title":"The power of scale for parameter-efficient prompt tuning. arXiv","author":"Lester Brian","year":"2021","unstructured":"Brian Lester, Rami Al-Rfou, and Noah Constant. 2021. The power of scale for parameter-efficient prompt tuning. arXiv (2021)."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"crossref","unstructured":"Liunian Harold Li Pengchuan Zhang Haotian Zhang Jianwei Yang Chunyuan Li Yiwu Zhong Lijuan Wang Lu Yuan Lei Zhang Jenq-Neng Hwang et al. 2022. Grounded language-image pre-training. In CVPR.","DOI":"10.1109\/CVPR52688.2022.01069"},{"key":"e_1_3_2_1_34_1","volume-title":"Referring transformer: A one-step approach to multi-task visual grounding. NeurIPS","author":"Li Muchen","year":"2021","unstructured":"Muchen Li and Leonid Sigal. 2021. Referring transformer: A one-step approach to multi-task visual grounding. NeurIPS (2021)."},{"key":"e_1_3_2_1_35_1","unstructured":"Ruiyu Li Kaican Li Yi-Chun Kuo Michelle Shu Xiaojuan Qi Xiaoyong Shen and Jiaya Jia. 2018. Referring image segmentation via recurrent refinement networks. In CVPR."},{"key":"e_1_3_2_1_36_1","unstructured":"Yanwei Li Xinze Chen Zheng Zhu Lingxi Xie Guan Huang Dalong Du and Xingang Wang. 2019. Attention-guided unified network for panoptic segmentation. In CVPR."},{"key":"e_1_3_2_1_37_1","volume-title":"Mail: A unified mask-image-language trimodal network for referring image segmentation. arXiv","author":"Li Zizhang","year":"2021","unstructured":"Zizhang Li, Mengmeng Wang, Jianbiao Mei, and Yong Liu. 2021. Mail: A unified mask-image-language trimodal network for referring image segmentation. arXiv (2021)."},{"key":"e_1_3_2_1_38_1","unstructured":"Tsung-Yi Lin Michael Maire Serge Belongie James Hays Pietro Perona Deva Ramanan Piotr Doll\u00e1r and C Lawrence Zitnick. 2014. Microsoft coco: Common objects in context. In ECCV."},{"key":"e_1_3_2_1_39_1","volume-title":"GRES: Generalized Referring Expression Segmentation. In CVPR.","author":"Liu Chang","year":"2023","unstructured":"Chang Liu, Henghui Ding, and Xudong Jiang. 2023. GRES: Generalized Referring Expression Segmentation. In CVPR."},{"key":"e_1_3_2_1_40_1","volume-title":"GRES: Generalized referring expression segmentation. In CVPR.","author":"Liu Chang","year":"2023","unstructured":"Chang Liu, Henghui Ding, and Xudong Jiang. 2023. GRES: Generalized referring expression segmentation. In CVPR."},{"key":"e_1_3_2_1_41_1","volume-title":"Instance-specific feature propagation for referring segmentation. ACM MM","author":"Liu Chang","year":"2022","unstructured":"Chang Liu, Xudong Jiang, and Henghui Ding. 2022. Instance-specific feature propagation for referring segmentation. ACM MM (2022)."},{"key":"e_1_3_2_1_42_1","unstructured":"Chenxi Liu Zhe Lin Xiaohui Shen Jimei Yang Xin Lu and Alan Yuille. 2017. Recurrent multimodal interaction for referring image segmentation. In ICCV."},{"key":"e_1_3_2_1_43_1","unstructured":"Daqing Liu Hanwang Zhang Feng Wu and Zheng-Jun Zha. 2019. Learning to assemble neural module tree networks for visual grounding. In ICCV."},{"key":"e_1_3_2_1_44_1","volume-title":"Vijay Mahadevan, and R Manmatha.","author":"Liu Jiang","year":"2023","unstructured":"Jiang Liu, Hui Ding, Zhaowei Cai, Yuting Zhang, Ravi Kumar Satzoda, Vijay Mahadevan, and R Manmatha. 2023. PolyFormer: Referring image segmentation as sequential polygon generation. In CVPR."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"crossref","unstructured":"Jiang Liu Hui Ding Zhaowei Cai Yuting Zhang Ravi Kumar Satzoda Vijay Mahadevan and R. Manmatha. 2023 d. PolyFormer: Referring Image Segmentation As Sequential Polygon Generation. In CVPR.","DOI":"10.1109\/CVPR52729.2023.01789"},{"key":"e_1_3_2_1_46_1","volume-title":"2023 e. Pre-train, prompt, and predict: A systematic survey of prompting methods in natural language processing. ACM CSUR","author":"Liu Pengfei","year":"2023","unstructured":"Pengfei Liu, Weizhe Yuan, Jinlan Fu, Zhengbao Jiang, Hiroaki Hayashi, and Graham Neubig. 2023 e. Pre-train, prompt, and predict: A systematic survey of prompting methods in natural language processing. ACM CSUR (2023)."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"crossref","unstructured":"Sun-Ao Liu Yiheng Zhang Zhaofan Qiu Hongtao Xie Yongdong Zhang and Ting Yao. 2023 f. CARIS: Context-aware referring image segmentation. In ACM MM. 779--788.","DOI":"10.1145\/3581783.3612117"},{"key":"e_1_3_2_1_48_1","volume-title":"Zhengxiao Du, Zhilin Yang, and Jie Tang.","author":"Liu Xiao","year":"2021","unstructured":"Xiao Liu, Kaixuan Ji, Yicheng Fu, Weng Lam Tam, Zhengxiao Du, Zhilin Yang, and Jie Tang. 2021. P-tuning v2: Prompt tuning can be comparable to fine-tuning universally across scales and tasks. arXiv (2021)."},{"key":"e_1_3_2_1_49_1","volume-title":"2023 g. Universal Segmentation at Arbitrary Granularity with Language Instruction. arXiv","author":"Liu Yong","year":"2023","unstructured":"Yong Liu, Cairong Zhang, Yitong Wang, Jiahao Wang, Yujiu Yang, and Yansong Tang. 2023 g. Universal Segmentation at Arbitrary Granularity with Language Instruction. arXiv (2023)."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"crossref","unstructured":"Ze Liu Yutong Lin Yue Cao Han Hu Yixuan Wei Zheng Zhang Stephen Lin and Baining Guo. 2021. Swin transformer: Hierarchical vision transformer using shifted windows. In ICCV.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"crossref","unstructured":"Jonathan Long Evan Shelhamer and Trevor Darrell. 2015. Fully convolutional networks for semantic segmentation. In CVPR.","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"e_1_3_2_1_52_1","volume-title":"Towards efficient visual adaption via structural re-parameterization. arXiv","author":"Luo Gen","year":"2023","unstructured":"Gen Luo, Minglang Huang, Yiyi Zhou, Xiaoshuai Sun, Guannan Jiang, Zhiyu Wang, and Rongrong Ji. 2023. Towards efficient visual adaption via structural re-parameterization. arXiv (2023)."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"crossref","unstructured":"Gen Luo Yiyi Zhou Rongrong Ji Xiaoshuai Sun Jinsong Su Chia-Wen Lin and Qi Tian. 2020. Cascade grouped attention network for referring expression segmentation. In ACM MM.","DOI":"10.1145\/3394171.3414006"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"crossref","unstructured":"Gen Luo Yiyi Zhou Xiaoshuai Sun Liujuan Cao Chenglin Wu Cheng Deng and Rongrong Ji. 2020. Multi-task collaborative network for joint referring expression comprehension and segmentation. In CVPR.","DOI":"10.1109\/CVPR42600.2020.01005"},{"key":"e_1_3_2_1_55_1","unstructured":"Lufan Ma Tiancai Wang Bin Dong Jiangpeng Yan Xiu Li and Xiangyu Zhang. 2021. Implicit feature refinement for instance segmentation. In ACM MM."},{"key":"e_1_3_2_1_56_1","unstructured":"Junhua Mao Jonathan Huang Alexander Toshev Oana Camburu Alan L Yuille and Kevin Murphy. 2016. Generation and comprehension of unambiguous object descriptions. In CVPR."},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"crossref","unstructured":"Edgar Margffoy-Tuay Juan C P\u00e9rez Emilio Botero and Pablo Arbel\u00e1ez. 2018. Dynamic multimodal instance segmentation guided by natural language queries. In ECCV.","DOI":"10.1007\/978-3-030-01252-6_39"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"crossref","unstructured":"Varun K Nagaraja Vlad I Morariu and Larry S Davis. 2016. Modeling context between objects for referring expression understanding. In ECCV.","DOI":"10.1007\/978-3-319-46493-0_48"},{"key":"e_1_3_2_1_59_1","unstructured":"Long Ouyang Jeffrey Wu Xu Jiang Diogo Almeida Carroll Wainwright Pamela Mishkin Chong Zhang Sandhini Agarwal Katarina Slama Alex Ray et al. 2022. Training language models to follow instructions with human feedback. NeurIPS (2022)."},{"key":"e_1_3_2_1_60_1","volume-title":"Shameema Sikder, S Swaroop Vedula, and Vishal M Patel.","author":"Paranjape Jay N","year":"2023","unstructured":"Jay N Paranjape, Nithin Gopalakrishnan Nair, Shameema Sikder, S Swaroop Vedula, and Vishal M Patel. 2023. Adaptivesam: Towards efficient tuning of sam for surgical scene segmentation. arXiv (2023)."},{"key":"e_1_3_2_1_61_1","unstructured":"Quan Quan Fenghe Tang Zikang Xu Heqin Zhu and S Kevin Zhou. 2024. Slide-SAM: Medical SAM Meets Sliding Window. In MIDL."},{"key":"e_1_3_2_1_62_1","volume-title":"U-net: Convolutional networks for biomedical image segmentation. In MICCAI.","author":"Ronneberger Olaf","year":"2015","unstructured":"Olaf Ronneberger, Philipp Fischer, and Thomas Brox. 2015. U-net: Convolutional networks for biomedical image segmentation. In MICCAI."},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"crossref","unstructured":"Haonan Shi Wenwen Pan Zhou Zhao Mingmin Zhang and Fei Wu. 2023. Unsupervised Domain Adaptation for Referring Semantic Segmentation. In ACM MM.","DOI":"10.1145\/3581783.3611879"},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"crossref","unstructured":"Zhi Tian Chunhua Shen and Hao Chen. 2020. Conditional convolutions for instance segmentation. In ECCV.","DOI":"10.1007\/978-3-030-58452-8_17"},{"key":"e_1_3_2_1_65_1","volume-title":"Llama: Open and efficient foundation language models. arXiv","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Thibaut Lavril, Gautier Izacard, Xavier Martinet, Marie-Anne Lachaux, Timoth\u00e9e Lacroix, Baptiste Rozi\u00e8re, Naman Goyal, Eric Hambro, Faisal Azhar, et al. 2023. Llama: Open and efficient foundation language models. arXiv (2023)."},{"key":"e_1_3_2_1_66_1","volume-title":"Attention is all you need. NeurIPS","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. NeurIPS (2017)."},{"key":"e_1_3_2_1_67_1","volume-title":"Solo: Segmenting objects by locations. In ECCV.","author":"Wang Xinlong","year":"2020","unstructured":"Xinlong Wang, Tao Kong, Chunhua Shen, Yuning Jiang, and Lei Li. 2020. Solo: Segmenting objects by locations. In ECCV."},{"key":"e_1_3_2_1_68_1","volume-title":"Solov2: Dynamic and fast instance segmentation. NeurIPS","author":"Wang Xinlong","year":"2020","unstructured":"Xinlong Wang, Rufeng Zhang, Tao Kong, Lei Li, and Chunhua Shen. 2020. Solov2: Dynamic and fast instance segmentation. NeurIPS (2020)."},{"key":"e_1_3_2_1_69_1","volume-title":"Cris: Clip-driven referring image segmentation. In CVPR.","author":"Wang Zhaoqing","year":"2022","unstructured":"Zhaoqing Wang, Yu Lu, Qiang Li, Xunqiang Tao, Yandong Guo, Mingming Gong, and Tongliang Liu. 2022. Cris: Clip-driven referring image segmentation. In CVPR."},{"key":"e_1_3_2_1_70_1","volume-title":"Brian Lester, Nan Du, Andrew M Dai, and Quoc V Le.","author":"Wei Jason","year":"2021","unstructured":"Jason Wei, Maarten Bosma, Vincent Y Zhao, Kelvin Guu, Adams Wei Yu, Brian Lester, Nan Du, Andrew M Dai, and Quoc V Le. 2021. Finetuned language models are zero-shot learners. arXiv (2021)."},{"key":"e_1_3_2_1_71_1","volume-title":"Medical sam adapter: Adapting segment anything model for medical image segmentation. arXiv","author":"Wu Junde","year":"2023","unstructured":"Junde Wu, Rao Fu, Huihui Fang, Yuanpei Liu, Zhaowei Wang, Yanwu Xu, Yueming Jin, and Tal Arbel. 2023. Medical sam adapter: Adapting segment anything model for medical image segmentation. arXiv (2023)."},{"key":"e_1_3_2_1_72_1","unstructured":"Jiannan Wu Yi Jiang Bin Yan Huchuan Lu Zehuan Yuan and Ping Luo. 2023. Segment Every Reference Object in Spatial and Temporal Spaces. In ICCV."},{"key":"e_1_3_2_1_73_1","volume-title":"2023 d. Infoprompt: Information-theoretic soft prompt tuning for natural language understanding. NeurIPS","author":"Wu Junda","year":"2023","unstructured":"Junda Wu, Tong Yu, Rui Wang, Zhao Song, Ruiyi Zhang, Handong Zhao, Chaochao Lu, Shuai Li, and Ricardo Henao. 2023 d. Infoprompt: Information-theoretic soft prompt tuning for natural language understanding. NeurIPS (2023)."},{"key":"e_1_3_2_1_74_1","volume-title":"Approximated Prompt Tuning for Vision-Language Pre-trained Models. arXiv","author":"Wu Qiong","year":"2023","unstructured":"Qiong Wu, Shubin Huang, Yiyi Zhou, Pingyang Dai, Annan Shu, Guannan Jiang, and Rongrong Ji. 2023. Approximated Prompt Tuning for Vision-Language Pre-trained Models. arXiv (2023)."},{"key":"e_1_3_2_1_75_1","volume-title":"CAT-SAM: Conditional Tuning Network for Few-Shot Adaptation of Segmentation Anything Model. arXiv","author":"Xiao Aoran","year":"2024","unstructured":"Aoran Xiao, Weihao Xuan, Heli Qi, Yun Xing, Ruijie Ren, Xiaoqin Zhang, and Shijian Lu. 2024. CAT-SAM: Conditional Tuning Network for Few-Shot Adaptation of Segmentation Anything Model. arXiv (2024)."},{"key":"e_1_3_2_1_76_1","volume-title":"MaskSAM: Towards Auto-prompt SAM with Mask Classification for Medical Image Segmentation. arXiv","author":"Xie Bin","year":"2024","unstructured":"Bin Xie, Hao Tang, Bin Duan, Dawen Cai, and Yan Yan. 2024. MaskSAM: Towards Auto-prompt SAM with Mask Classification for Medical Image Segmentation. arXiv (2024)."},{"key":"e_1_3_2_1_77_1","volume-title":"Edit everything: A text-guided generative system for images editing. arXiv","author":"Xie Defeng","year":"2023","unstructured":"Defeng Xie, Ruichen Wang, Jian Ma, Chen Chen, Haonan Lu, Dong Yang, Fobo Shi, and Xiaodong Lin. 2023. Edit everything: A text-guided generative system for images editing. arXiv (2023)."},{"key":"e_1_3_2_1_78_1","volume-title":"SegFormer: Simple and efficient design for semantic segmentation with transformers. NeurIPS","author":"Xie Enze","year":"2021","unstructured":"Enze Xie, Wenhai Wang, Zhiding Yu, Anima Anandkumar, Jose M Alvarez, and Ping Luo. 2021. SegFormer: Simple and efficient design for semantic segmentation with transformers. NeurIPS (2021)."},{"key":"e_1_3_2_1_79_1","volume-title":"Xindi Shang, Zehuan Yuan, Ying Sun, and Jun Liu.","author":"Xu Li","year":"2023","unstructured":"Li Xu, Mark He Huang, Xindi Shang, Zehuan Yuan, Ying Sun, and Jun Liu. 2023. Meta Compositional Referring Expression Segmentation. In CVPR."},{"key":"e_1_3_2_1_80_1","unstructured":"Zunnan Xu Zhihong Chen Yong Zhang Yibing Song Xiang Wan and Guanbin Li. 2023. Bridging vision and language encoders: Parameter-efficient tuning for referring image segmentation. In ICCV."},{"key":"e_1_3_2_1_81_1","doi-asserted-by":"crossref","unstructured":"Bin Yan Yi Jiang Jiannan Wu Dong Wang Ping Luo Zehuan Yuan and Huchuan Lu. 2023. Universal instance perception as object discovery and retrieval. In CVPR.","DOI":"10.1109\/CVPR52729.2023.01471"},{"key":"e_1_3_2_1_82_1","doi-asserted-by":"crossref","unstructured":"Sibei Yang Meng Xia Guanbin Li Hong-Yu Zhou and Yizhou Yu. 2021. Bottom-up shift and reasoning for referring image segmentation. In CVPR.","DOI":"10.1109\/CVPR46437.2021.01111"},{"key":"e_1_3_2_1_83_1","volume-title":"Lavt: Language-aware vision transformer for referring image segmentation. In CVPR.","author":"Yang Zhao","year":"2022","unstructured":"Zhao Yang, Jiaqi Wang, Yansong Tang, Kai Chen, Hengshuang Zhao, and Philip HS Torr. 2022. Lavt: Language-aware vision transformer for referring image segmentation. In CVPR."},{"key":"e_1_3_2_1_84_1","doi-asserted-by":"crossref","unstructured":"Zhao Yang Jiaqi Wang Yansong Tang Kai Chen Hengshuang Zhao and Philip HS Torr. 2023. Semantics-aware dynamic localization and refinement for referring image segmentation. In AAAI.","DOI":"10.1609\/aaai.v37i3.25428"},{"key":"e_1_3_2_1_85_1","volume-title":"Gaussian grouping: Segment and edit anything in 3d scenes. arXiv","author":"Ye Mingqiao","year":"2023","unstructured":"Mingqiao Ye, Martin Danelljan, Fisher Yu, and Lei Ke. 2023. Gaussian grouping: Segment and edit anything in 3d scenes. arXiv (2023)."},{"key":"e_1_3_2_1_86_1","unstructured":"Licheng Yu Patrick Poirson Shan Yang Alexander C Berg and Tamara L Berg. 2016. Modeling context in referring expressions. In ECCV."},{"key":"e_1_3_2_1_87_1","doi-asserted-by":"crossref","unstructured":"Pingping Zhang Tianyu Yan Yang Liu and Huchuan Lu. 2024. Fantastic Animals and Where to Find Them: Segment Any Marine Animal with Dual SAM. In CVPR.","DOI":"10.1109\/CVPR52733.2024.00249"},{"key":"e_1_3_2_1_88_1","doi-asserted-by":"crossref","unstructured":"Xin Zhang Yu Liu Yuming Lin Qingmin Liao and Yong Li. 2024. UV-SAM: Adapting segment anything model for urban village identification. In AAAI.","DOI":"10.1609\/aaai.v38i20.30260"},{"key":"e_1_3_2_1_89_1","doi-asserted-by":"crossref","unstructured":"Wenliang Zhao Yongming Rao Zuyan Liu Benlin Liu Jie Zhou and Jiwen Lu. 2023. Unleashing text-to-image diffusion models for visual perception. In ICCV.","DOI":"10.1109\/ICCV51070.2023.00527"},{"key":"e_1_3_2_1_90_1","volume-title":"Chen Change Loy, and Ziwei Liu","author":"Zhou Kaiyang","year":"2022","unstructured":"Kaiyang Zhou, Jingkang Yang, Chen Change Loy, and Ziwei Liu. 2022. Conditional prompt learning for vision-language models. In CVPR."},{"key":"e_1_3_2_1_91_1","volume-title":"Chen Change Loy, and Ziwei Liu","author":"Zhou Kaiyang","year":"2022","unstructured":"Kaiyang Zhou, Jingkang Yang, Chen Change Loy, and Ziwei Liu. 2022. Learning to prompt for vision-language models. IJCV (2022)."},{"key":"e_1_3_2_1_92_1","volume-title":"Seqtr: A simple yet universal network for visual grounding. In ECCV.","author":"Zhu Chaoyang","year":"2022","unstructured":"Chaoyang Zhu, Yiyi Zhou, Yunhang Shen, Gen Luo, Xingjia Pan, Mingbao Lin, Chao Chen, Liujuan Cao, Xiaoshuai Sun, and Rongrong Ji. 2022. Seqtr: A simple yet universal network for visual grounding. In ECCV."},{"key":"e_1_3_2_1_93_1","volume-title":"Segment everything everywhere all at once. NeurIPS","author":"Zou Xueyan","year":"2023","unstructured":"Xueyan Zou, Jianwei Yang, Hao Zhang, Feng Li, Linjie Li, Jianfeng Wang, Lijuan Wang, Jianfeng Gao, and Yong Jae Lee. 2023. Segment everything everywhere all at once. NeurIPS (2023)."},{"key":"e_1_3_2_1_94_1","unstructured":"Xinyan Zu Haiyang Yu Bin Li and Xiangyang Xue. 2023. Weakly-supervised text instance segmentation. In ACM MM."}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3680571","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3680571","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:56Z","timestamp":1750295876000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3680571"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":94,"alternative-id":["10.1145\/3664647.3680571","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3680571","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}