{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,21]],"date-time":"2026-05-21T16:57:22Z","timestamp":1779382642694,"version":"3.53.1"},"publisher-location":"New York, NY, USA","reference-count":53,"publisher":"ACM","funder":[{"name":"Shanghai Natural Science Foundation","award":["No. 25ZR1401028"],"award-info":[{"award-number":["No. 25ZR1401028"]}]},{"DOI":"10.13039\/501100003399","name":"Science and Technology Commission of Shanghai Municipality","doi-asserted-by":"publisher","award":["No. 23511100602; No. 21511104506"],"award-info":[{"award-number":["No. 23511100602; No. 21511104506"]}],"id":[{"id":"10.13039\/501100003399","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755166","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:37:21Z","timestamp":1761377841000},"page":"362-371","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["Text-Promptable Propagation for Referring Medical Image Sequence Segmentation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-6016-8452","authenticated-orcid":false,"given":"Runtian","family":"Yuan","sequence":"first","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9824-5852","authenticated-orcid":false,"given":"Mohan","family":"Chen","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8208-2067","authenticated-orcid":false,"given":"Jilan","family":"Xu","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9508-4950","authenticated-orcid":false,"given":"Ling","family":"Zhou","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-4848-4583","authenticated-orcid":false,"given":"Qingqiu","family":"Li","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7993-7223","authenticated-orcid":false,"given":"Yuejie","family":"Zhang","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4747-0574","authenticated-orcid":false,"given":"Rui","family":"Feng","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7561-0143","authenticated-orcid":false,"given":"Tao","family":"Zhang","sequence":"additional","affiliation":[{"name":"Shanghai University of Finance and Economics, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2947-7780","authenticated-orcid":false,"given":"Shang","family":"Gao","sequence":"additional","affiliation":[{"name":"Deakin University, Geelong, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","unstructured":"H. J. W. L. Aerts L. Wee E. Rios Velazquez R. T. H. Leijenaar C. Parmar P. Grossmann S. Carvalho J. Bussink R. Monshouwer B. Haibe-Kains D. Rietveld F. Hoebers M. M. Rietbergen C. R. Leemans A. Dekker J. Quackenbush R. J. Gillies and P. Lambin. 2019. Data From NSCLC-Radiomics (Version 4). Data set. doi:10.7937\/K9\/TCIA.2015.PF0M9REI","DOI":"10.7937\/K9\/TCIA.2015.PF0M9REI"},{"key":"e_1_3_2_1_2_1","unstructured":"Ujjwal Baid Satyam Ghodasara Suyash Mohan Michel Bilello Evan Calabrese Errol Colak Keyvan Farahani Jayashree Kalpathy-Cramer Felipe C Kitamura Sarthak Pati et al. 2021. The rsna-asnr-miccai brats 2021 benchmark on brain tumor segmentation and radiogenomic classification. arXiv preprint arXiv:2107.02314 (2021)."},{"key":"e_1_3_2_1_3_1","volume-title":"Advancing the cancer genome atlas glioma MRI collections with expert segmentation labels and radiomic features. Scientific data","author":"Bakas Spyridon","year":"2017","unstructured":"Spyridon Bakas, Hamed Akbari, Aristeidis Sotiras, Michel Bilello, Martin Rozycki, Justin S Kirby, John B Freymann, Keyvan Farahani, and Christos Davatzikos. 2017. Advancing the cancer genome atlas glioma MRI collections with expert segmentation labels and radiomic features. Scientific data, Vol. 4, 1 (2017), 1-13."},{"key":"e_1_3_2_1_4_1","volume-title":"WM-DOVA maps for accurate polyp highlighting in colonoscopy: Validation vs. saliency maps from physicians. Computerized medical imaging and graphics","author":"Bernal Jorge","year":"2015","unstructured":"Jorge Bernal, F Javier S\u00e1nchez, Gloria Fern\u00e1ndez-Esparrach, Debora Gil, Cristina Rodr\u00edguez, and Fernando Vilari no. 2015. WM-DOVA maps for accurate polyp highlighting in colonoscopy: Validation vs. saliency maps from physicians. Computerized medical imaging and graphics, Vol. 43 (2015), 99-111."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2012.03.002"},{"key":"e_1_3_2_1_6_1","volume-title":"Miguel Angel Gonzalez Ballester, et al","author":"Bernard Olivier","year":"2018","unstructured":"Olivier Bernard, Alain Lalande, Clement Zotti, Frederick Cervenansky, Xin Yang, Pheng-Ann Heng, Irem Cetin, Karim Lekadir, Oscar Camara, Miguel Angel Gonzalez Ballester, et al., 2018. Deep learning techniques for automatic MRI cardiac multi-structures segmentation and diagnosis: is the problem solved? IEEE transactions on medical imaging, Vol. 37, 11 (2018), 2514-2525."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.media.2022.102680"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00493"},{"key":"e_1_3_2_1_9_1","volume-title":"Visual-Textual Matching Attention for Lesion Segmentation in Chest Images. In International Conference on Medical Image Computing and Computer-Assisted Intervention. Springer, 702-711","author":"Bui Phuoc-Nguyen","year":"2024","unstructured":"Phuoc-Nguyen Bui, Duc-Tai Le, and Hyunseung Choo. 2024. Visual-Textual Matching Attention for Lesion Segmentation in Chest Images. In International Conference on Medical Image Computing and Computer-Assisted Intervention. Springer, 702-711."},{"key":"e_1_3_2_1_10_1","volume-title":"European conference on computer vision. Springer, 205-218","author":"Cao Hu","year":"2022","unstructured":"Hu Cao, Yueyue Wang, Joy Chen, Dongsheng Jiang, Xiaopeng Zhang, Qi Tian, and Manning Wang. 2022. Swin-unet: Unet-like pure transformer for medical image segmentation. In European conference on computer vision. Springer, 205-218."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"Kai Cao Yingda Xia Jiawen Yao Xu Han Lukas Lambert Tingting Zhang Wei Tang Gang Jin Hui Jiang Xu Fang et al. 2023. Large-scale pancreatic cancer detection via non-contrast CT and deep learning. Nature medicine Vol. 29 12 (2023) 3033-3043.","DOI":"10.1038\/s41591-023-02640-w"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMI.2018.2890510"},{"key":"e_1_3_2_1_13_1","volume-title":"Transunet: Transformers make strong encoders for medical image segmentation. arXiv preprint arXiv:2102.04306","author":"Chen Jieneng","year":"2021","unstructured":"Jieneng Chen, Yongyi Lu, Qihang Yu, Xiangde Luo, Ehsan Adeli, Yan Wang, Le Lu, Alan L Yuille, and Yuyin Zhou. 2021. Transunet: Transformers make strong encoders for medical image segmentation. arXiv preprint arXiv:2102.04306 (2021)."},{"key":"e_1_3_2_1_14_1","first-page":"424","volume-title":"International Conference on Medical Image Computing and Computer-Assisted Intervention, Part II 19","author":"\u00c7i\u00e7ek \u00d6zg\u00fcn","year":"2016","unstructured":"\u00d6zg\u00fcn \u00c7i\u00e7ek, Ahmed Abdulkadir, Soeren S Lienkamp, Thomas Brox, and Olaf Ronneberger. 2016. 3D U-Net: learning dense volumetric segmentation from sparse annotation. In International Conference on Medical Image Computing and Computer-Assisted Intervention, Part II 19. Springer, 424-432."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00624"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV51458.2022.00181"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"crossref","unstructured":"Nicholas Heller Fabian Isensee Klaus H Maier-Hein Xiaoshuai Hou Chunmei Xie Fengyi Li Yang Nan Guangrui Mu Zhiyong Lin Miofei Han et al. 2021. The state of the art in kidney and kidney tumor segmentation in contrast-enhanced CT imaging: Results of the KiTS19 challenge. Medical image analysis Vol. 67 (2021) 101821.","DOI":"10.1016\/j.media.2020.101821"},{"key":"e_1_3_2_1_19_1","unstructured":"Nicholas Heller Fabian Isensee Dasha Trofimova Resha Tejpaul Zhongchen Zhao Huai Chen Lisheng Wang Alex Golts Daniel Khapun Daniel Shats et al. 2023. The kits21 challenge: Automatic segmentation of kidneys renal tumors and renal cysts in corticomedullary-phase ct. arXiv preprint arXiv:2307.01984 (2023)."},{"key":"e_1_3_2_1_20_1","volume-title":"Jens Petersen, and Klaus H Maier-Hein.","author":"Isensee Fabian","year":"2021","unstructured":"Fabian Isensee, Paul F Jaeger, Simon AA Kohl, Jens Petersen, and Klaus H Maier-Hein. 2021. nnU-Net: a self-configuring method for deep learning-based biomedical image segmentation. Nature methods, Vol. 18, 2 (2021), 203-211."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-87193-2_14"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"crossref","unstructured":"Hongxu Jiang Muhammad Imran Preethika Muralidharan Anjali Patel Jake Pensa Muxuan Liang Tarik Benidir Joseph R Grajo Jason P Joseph Russell Terry et al. 2024. MicroSegNet: A deep learning approach for prostate segmentation on micro-ultrasound images. Computerized Medical Imaging and Graphics (2024) 102326.","DOI":"10.1016\/j.compmedimag.2024.102326"},{"key":"e_1_3_2_1_23_1","volume-title":"Proc. MICCAI Multi-Atlas Labeling Beyond Cranial Vault-Workshop Challenge","volume":"5","author":"Landman Bennett","year":"2015","unstructured":"Bennett Landman, Zhoubing Xu, J Igelsias, Martin Styner, Thomas Langerak, and Arno Klein. 2015. Miccai multi-atlas labeling beyond the cranial vault-workshop and challenge. In Proc. MICCAI Multi-Atlas Labeling Beyond Cranial Vault-Workshop Challenge, Vol. 5. 12."},{"key":"e_1_3_2_1_24_1","volume-title":"Pierre-Marc Jodoin, Thomas Grenier, et al.","author":"Leclerc Sarah","year":"2019","unstructured":"Sarah Leclerc, Erik Smistad, Joao Pedrosa, Andreas \u00d8stvik, Frederic Cervenansky, Florian Espinosa, Torvald Espeland, Erik Andreas Rye Berg, Pierre-Marc Jodoin, Thomas Grenier, et al., 2019. Deep learning for segmentation using an open large-scale dataset in 2D echocardiography. IEEE transactions on medical imaging, Vol. 38, 9 (2019), 2198-2210."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72120-5_8"},{"key":"e_1_3_2_1_26_1","volume-title":"Lvit: language meets vision transformer in medical image segmentation","author":"Li Zihan","year":"2023","unstructured":"Zihan Li, Yunxiang Li, Qingde Li, Puyang Wang, Dazhou Guo, Le Lu, Dakai Jin, You Zhang, and Qingqi Hong. 2023. Lvit: language meets vision transformer in medical image segmentation. IEEE transactions on medical imaging (2023)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-43898-1_48"},{"key":"e_1_3_2_1_28_1","unstructured":"Shilong Liu Zhaoyang Zeng Tianhe Ren Feng Li Hao Zhang Jie Yang Chunyuan Li Jianwei Yang Hang Su Jun Zhu et al. 2023. Grounding dino: Marrying dino with grounded pre-training for open-set object detection. arXiv preprint arXiv:2303.05499 (2023)."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"e_1_3_2_1_31_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Luo Zhuoyan","year":"2024","unstructured":"Zhuoyan Luo, Yicheng Xiao, Yong Liu, Shuyan Li, Yitong Wang, Yansong Tang, Xiu Li, and Yujiu Yang. 2024. Soc: Semantic-assisted object cluster for referring video object segmentation. Advances in Neural Information Processing Systems, Vol. 36 (2024)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41467-024-44824-z"},{"key":"e_1_3_2_1_33_1","unstructured":"Bjoern H Menze Andras Jakab Stefan Bauer Jayashree Kalpathy-Cramer Keyvan Farahani Justin Kirby Yuliya Burren Nicole Porz Johannes Slotboom Roland Wiest et al. 2014. The multimodal brain tumor image segmentation benchmark (BRATS). IEEE transactions on medical imaging Vol. 34 10 (2014) 1993-2024."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","unstructured":"C. R. Meyer T. L. Chenevert C. J. Galb\u00e1n T. D. Johnson D. A. Hamstra A. Rehemtulla and B. D. Ross. 2015. RIDER Breast MRI. Data set. doi:10.7937\/K9\/TCIA.2015.H1SXNUXL","DOI":"10.7937\/K9\/TCIA.2015.H1SXNUXL"},{"key":"e_1_3_2_1_35_1","first-page":"565","article-title":"V-net: Fully convolutional neural networks for volumetric medical image segmentation. In 2016 fourth international conference on 3D vision (3DV)","author":"Milletari Fausto","year":"2016","unstructured":"Fausto Milletari, Nassir Navab, and Seyed-Ahmad Ahmadi. 2016. V-net: Fully convolutional neural networks for volumetric medical image segmentation. In 2016 fourth international conference on 3D vision (3DV). Ieee, 565-571.","journal-title":"Ieee"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMI.2022.3173669"},{"key":"e_1_3_2_1_37_1","volume-title":"M Jorge Cardoso, Hsiang-Chou Chen, et al.","author":"Petitjean Caroline","year":"2015","unstructured":"Caroline Petitjean, Maria A Zuluaga, Wenjia Bai, Jean-Nicolas Dacher, Damien Grosgeorge, J\u00e9r\u00f4me Caudron, Su Ruan, Ismail Ben Ayed, M Jorge Cardoso, Hsiang-Chou Chen, et al., 2015. Right ventricle segmentation from cardiac MRI: a collation study. Medical image analysis, Vol. 19, 1 (2015), 187-202."},{"key":"e_1_3_2_1_38_1","volume-title":"Nicolas Carion, Chao-Yuan Wu, Ross Girshick, Piotr Doll\u00e1r, and Christoph Feichtenhofer.","author":"Ravi Nikhila","year":"2024","unstructured":"Nikhila Ravi, Valentin Gabeur, Yuan-Ting Hu, Ronghang Hu, Chaitanya Ryali, Tengyu Ma, Haitham Khedr, Roman R\u00e4dle, Chloe Rolland, Laura Gustafson, Eric Mintun, Junting Pan, Kalyan Vasudev Alwala, Nicolas Carion, Chao-Yuan Wu, Ross Girshick, Piotr Doll\u00e1r, and Christoph Feichtenhofer. 2024. SAM 2: Segment Anything in Images and Videos. arXiv preprint arXiv:2408.00714 (2024)."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-24553-9_68"},{"key":"e_1_3_2_1_41_1","volume-title":"Urvos: Unified referring video object segmentation network with a large-scale benchmark. In Computer Vision-ECCV 2020: 16th European Conference","author":"Seo Seonguk","year":"2020","unstructured":"Seonguk Seo, Joon-Young Lee, and Bohyung Han. 2020. Urvos: Unified referring video object segmentation network with a large-scale benchmark. In Computer Vision-ECCV 2020: 16th European Conference, Glasgow, UK, August 23-28, 2020, Proceedings, Part XV 16. Springer, 208-223."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11548-013-0926-3"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jamcollsurg.2014.12.008"},{"key":"e_1_3_2_1_44_1","volume-title":"Automated polyp detection in colonoscopy videos using shape and context information","author":"Tajbakhsh Nima","year":"2015","unstructured":"Nima Tajbakhsh, Suryakanth R Gurudu, and Jianming Liang. 2015. Automated polyp detection in colonoscopy videos using shape and context information. IEEE transactions on medical imaging, Vol. 35, 2 (2015), 630-644."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_17"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00259"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00492"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"crossref","unstructured":"Zhaohan Xiong Qing Xia Zhiqiang Hu Ning Huang Cheng Bian Yefeng Zheng Sulaiman Vesal Nishant Ravikumar Andreas Maier Xin Yang et al. 2021. A global benchmark of algorithms for segmenting the left atrium from late gadolinium-enhanced cardiac magnetic resonance imaging. Medical image analysis Vol. 67 (2021) 101832.","DOI":"10.1016\/j.media.2020.101832"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i6.28465"},{"key":"e_1_3_2_1_50_1","volume-title":"Patterns","volume":"4","author":"Zhang Jiadong","year":"2023","unstructured":"Jiadong Zhang, Zhiming Cui, Zhenwei Shi, Yingjia Jiang, Zhiliang Zhang, Xiaoting Dai, Zhenlu Yang, Yuning Gu, Lei Zhou, Chu Han, et al., 2023. A robust and efficient AI assistant for breast tumor segmentation from DCE-MRI via a spatial-temporal framework. Patterns, Vol. 4, 9 (2023)."},{"key":"e_1_3_2_1_51_1","volume-title":"One model to rule them all: Towards universal segmentation for medical images with text prompts. arXiv preprint arXiv:2312.17183","author":"Zhao Ziheng","year":"2023","unstructured":"Ziheng Zhao, Yao Zhang, Chaoyi Wu, Xiaoman Zhang, Ya Zhang, Yanfeng Wang, and Weidi Xie. 2023. One model to rule them all: Towards universal segmentation for medical images with text prompts. arXiv preprint arXiv:2312.17183 (2023)."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-43901-8_69"},{"key":"e_1_3_2_1_53_1","volume-title":"Deformable DETR: Deformable Transformers for End-to-End Object Detection. In International Conference on Learning Representations.","author":"Zhu Xizhou","year":"2021","unstructured":"Xizhou Zhu, Weijie Su, Lewei Lu, Bin Li, Xiaogang Wang, and Jifeng Dai. 2021. Deformable DETR: Deformable Transformers for End-to-End Object Detection. In International Conference on Learning Representations."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755166","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T05:04:42Z","timestamp":1765343082000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755166"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":53,"alternative-id":["10.1145\/3746027.3755166","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755166","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}