{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T01:10:25Z","timestamp":1755825025369,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":66,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,30]]},"DOI":"10.1145\/3731715.3733274","type":"proceedings-article","created":{"date-parts":[[2025,6,25]],"date-time":"2025-06-25T18:31:04Z","timestamp":1750876264000},"page":"1700-1709","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["ALVG: Training High-Quality Multi-modal Fusion Modules for Visual Grounding with Attention Loss"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-7887-5231","authenticated-orcid":false,"given":"Sicheng","family":"Yang","sequence":"first","affiliation":[{"name":"Key Laboratory of Aerospace Information Security and Trusted Computing, Wuhan University, Wuhan, Hubei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8359-7062","authenticated-orcid":false,"given":"Rongwei","family":"Yu","sequence":"additional","affiliation":[{"name":"Key Laboratory of Aerospace Information Security and Trusted Computing, Wuhan University, Wuhan, Hubei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,6,30]]},"reference":[{"doi-asserted-by":"publisher","key":"e_1_3_2_1_1_1","DOI":"10.1007\/978-3-030-58452-8_13"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_2_1","DOI":"10.1609\/aaai.v35i2.16188"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_3_1","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"e_1_3_2_1_4_1","volume-title":"Mask grounding for referring image segmentation. arXiv preprint arXiv:2312.12198","author":"Chng Yong Xien","year":"2023","unstructured":"Yong Xien Chng, Henry Zheng, Yizeng Han, Xuchong Qiu, and Gao Huang. Mask grounding for referring image segmentation. arXiv preprint arXiv:2312.12198, 2023."},{"key":"e_1_3_2_1_5_1","volume-title":"Simvg: A simple framework for visual grounding with decoupled multi-modal fusion. arXiv preprint arXiv:2409.17531","author":"Dai Ming","year":"2024","unstructured":"Ming Dai, Lingfeng Yang, Yihao Xu, Zhenhua Feng, and Wankou Yang. Simvg: A simple framework for visual grounding with decoupled multi-modal fusion. arXiv preprint arXiv:2409.17531, 2024."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_6_1","DOI":"10.1109\/ICCV48922.2021.00179"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_7_1","DOI":"10.1109\/ICCV48922.2021.01601"},{"key":"e_1_3_2_1_8_1","volume-title":"An image is worth 16x16 words: Transformers for image recognition at scale. arXiv","author":"Dosovitskiy Alexey","year":"2020","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, et al. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv, 2020."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_9_1","DOI":"10.1109\/ICME52920.2022.9859880"},{"key":"e_1_3_2_1_10_1","volume-title":"Large-scale adversarial training for vision-and-language representation learning. Advances in Neural Information Processing Systems (NeurIPS), 33:6616--6628","author":"Gan Zhe","year":"2020","unstructured":"Zhe Gan, Yen-Chun Chen, Linjie Li, Chen Zhu, Yu Cheng, and Jingjing Liu. Large-scale adversarial training for vision-and-language representation learning. Advances in Neural Information Processing Systems (NeurIPS), 33:6616--6628, 2020."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_11_1","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_12_1","volume-title":"GREC: Generalized referring expression comprehension. arXiv","author":"He Shuting","year":"2023","unstructured":"Shuting He, Henghui Ding, Chang Liu, and Xudong Jiang. GREC: Generalized referring expression comprehension. arXiv, 2023."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_13_1","DOI":"10.1109\/TPAMI.2019.2911066"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_14_1","DOI":"10.1109\/CVPR.2017.470"},{"key":"e_1_3_2_1_15_1","volume-title":"Beyond one-to-one: Rethinking the referring image segmentation","author":"Hu Yutao","year":"2023","unstructured":"Yutao Hu, Qixiong Wang, Wenqi Shao, Enze Xie, Zhenguo Li, Jungong Han, and Ping Luo. Beyond one-to-one: Rethinking the referring image segmentation, 2023."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_16_1","DOI":"10.1109\/CVPR46437.2021.01661"},{"key":"e_1_3_2_1_17_1","volume-title":"Densely connected parameter-efficient tuning for referring image segmentation. arXiv preprint arXiv:2501.08580","author":"Huang Jiaqi","year":"2025","unstructured":"Jiaqi Huang, Zunnan Xu, Ting Liu, Yong Liu, Haonan Han, Kehong Yuan, and Xiu Li. Densely connected parameter-efficient tuning for referring image segmentation. arXiv preprint arXiv:2501.08580, 2025."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_18_1","DOI":"10.1109\/CVPR46437.2021.00973"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_19_1","DOI":"10.1109\/ICCV48922.2021.00180"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_20_1","DOI":"10.3115\/v1\/D14-1086"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_21_1","DOI":"10.18653\/v1\/2024.naacl-long.258"},{"key":"e_1_3_2_1_22_1","volume-title":"Lisa: Reasoning segmentation via large language model. arXiv preprint arXiv:2308.00692","author":"Lai Xin","year":"2023","unstructured":"Xin Lai, Zhuotao Tian, Yukang Chen, Yanwei Li, Yuhui Yuan, Shu Liu, and Jiaya Jia. Lisa: Reasoning segmentation via large language model. arXiv preprint arXiv:2308.00692, 2023."},{"key":"e_1_3_2_1_23_1","volume-title":"Referring transformer: A one-step approach to multi-task visual grounding. Advances in Neural Information Processing Systems (NeurIPS), 34","author":"Li Muchen","year":"2021","unstructured":"Muchen Li and Leonid Sigal. Referring transformer: A one-step approach to multi-task visual grounding. Advances in Neural Information Processing Systems (NeurIPS), 34, 2021."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_24_1","DOI":"10.1109\/CVPR42600.2020.01089"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_25_1","DOI":"10.1109\/ICCV.2017.324"},{"key":"e_1_3_2_1_26_1","first-page":"740","volume-title":"Proceedings of the European Conference on Computer Vision (ECCV)","author":"Lin Tsung-Yi","year":"2014","unstructured":"Tsung-Yi Lin, Michael Maire, Serge Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Doll\u00e1r, and C Lawrence Zitnick. Microsoft coco: Common objects in context. In Proceedings of the European Conference on Computer Vision (ECCV), pages 740--755, 2014."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_27_1","DOI":"10.1109\/CVPR52729.2023.02259"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_28_1","DOI":"10.1109\/ICCV.2019.00477"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_29_1","DOI":"10.1109\/CVPR52729.2023.01789"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_30_1","DOI":"10.1609\/aaai.v37i2.25261"},{"key":"e_1_3_2_1_31_1","volume-title":"Grounding dino: Marrying dino with grounded pre-training for open-set object detection. arXiv","author":"Liu Shilong","year":"2023","unstructured":"Shilong Liu, Zhaoyang Zeng, Tianhe Ren, Feng Li, Hao Zhang, Jie Yang, Chunyuan Li, Jianwei Yang, Hang Su, Jun Zhu, et al. Grounding dino: Marrying dino with grounded pre-training for open-set object detection. arXiv, 2023."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_32_1","DOI":"10.1109\/CVPR.2019.00205"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_33_1","DOI":"10.1109\/CVPR42600.2020.01005"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_34_1","DOI":"10.1109\/CVPR.2016.9"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_35_1","DOI":"10.1109\/3DV.2016.79"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_36_1","DOI":"10.1007\/978-3-319-46493-0_48"},{"key":"e_1_3_2_1_37_1","volume-title":"Flickr30k entities: Collecting region-to-phrase correspondences for richer image-to-sentence models. Internatioanl Journal of Computer Vision (IJCV), 123(1):74--93","author":"Plummer Bryan A.","year":"2017","unstructured":"Bryan A. Plummer, Liwei Wang, Christopher M. Cervantes, Juan C. Caicedo, Julia Hockenmaier, and Svetlana Lazebnik. Flickr30k entities: Collecting region-to-phrase correspondences for richer image-to-sentence models. Internatioanl Journal of Computer Vision (IJCV), 123(1):74--93, 2017."},{"key":"e_1_3_2_1_38_1","volume-title":"Yolov3: An incremental improvement. arXiv","author":"Redmon Joseph","year":"2018","unstructured":"Joseph Redmon and Ali Farhadi. Yolov3: An incremental improvement. arXiv, 2018."},{"key":"e_1_3_2_1_39_1","volume-title":"Faster r-cnn: Towards real-time object detection with region proposal networks","author":"Ren Shaoqing","year":"2016","unstructured":"Shaoqing Ren, Kaiming He, Ross Girshick, and Jian Sun. Faster r-cnn: Towards real-time object detection with region proposal networks. IEEE transactions on pattern analysis and machine intelligence, 39(6):1137--1149, 2016."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_40_1","DOI":"10.1109\/CVPR.2019.00075"},{"key":"e_1_3_2_1_41_1","volume-title":"Dynamic mdetr: A dynamic multimodal transformer decoder for visual grounding","author":"Shi Fengyuan","year":"2023","unstructured":"Fengyuan Shi, Ruopeng Gao, Weilin Huang, and Limin Wang. Dynamic mdetr: A dynamic multimodal transformer decoder for visual grounding. IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI), 2023."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_42_1","DOI":"10.1109\/CVPR52729.2023.02257"},{"key":"e_1_3_2_1_43_1","first-page":"4965","volume-title":"International Conference on Machine Learning","author":"Trinh Trieu","year":"2018","unstructured":"Trieu Trinh, Andrew Dai, Thang Luong, and Quoc Le. Learning longer-term dependencies in rnns with auxiliary losses. In International Conference on Machine Learning, pages 4965--4974. PMLR, 2018."},{"key":"e_1_3_2_1_44_1","volume-title":"Attention is all you need. Advances in Neural Information Processing Systems (NeurIPS), 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. Attention is all you need. Advances in Neural Information Processing Systems (NeurIPS), 30, 2017."},{"key":"e_1_3_2_1_45_1","volume-title":"Unifying architectures, tasks, and modalities through a simple sequence-to-sequence learning framework. arXiv","author":"Wang Peng","year":"2022","unstructured":"Peng Wang, An Yang, Rui Men, Junyang Lin, Shuai Bai, Zhikang Li, Jianxin Ma, Chang Zhou, Jingren Zhou, and Hongxia Yang. Unifying architectures, tasks, and modalities through a simple sequence-to-sequence learning framework. arXiv, 2022."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_46_1","DOI":"10.1109\/CVPR52729.2023.01838"},{"key":"e_1_3_2_1_47_1","volume-title":"The Twelfth International Conference on Learning Representations","author":"Wang Yaoming","year":"2023","unstructured":"Yaoming Wang, Jin Li, XIAOPENG ZHANG, Bowen Shi, Chenglin Li, Wenrui Dai, Hongkai Xiong, and Qi Tian. Barleria: An efficient tuning framework for referring image segmentation. In The Twelfth International Conference on Learning Representations, 2023."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_48_1","DOI":"10.1109\/CVPR52688.2022.01139"},{"key":"e_1_3_2_1_49_1","volume-title":"Hyperseg: Towards universal visual segmentation with large language model. arXiv preprint arXiv:2411.17606","author":"Wei Cong","year":"2024","unstructured":"Cong Wei, Yujie Zhong, Haoxian Tan, Yong Liu, Zheng Zhao, Jie Hu, and Yujiu Yang. Hyperseg: Towards universal visual segmentation with large language model. arXiv preprint arXiv:2411.17606, 2024."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_50_1","DOI":"10.1109\/CVPR52733.2024.00370"},{"key":"e_1_3_2_1_51_1","volume-title":"International Conference on Machine Learning (ICML)","author":"Xu Haiyang","year":"2023","unstructured":"Haiyang Xu, Qinghao Ye, Mingshi Yan, Yaya Shi, Jiabo Ye, Yuanhong Xu, Chenliang Li, Bin Bi, Qiuchen Qian, Wei Wang, Guohai Xu, Ji Zhang, Songfang Huang, Feiran Huang, and Jingren Zhou. mplug-2: A modularized multi-modal foundation model across text, image and video. In International Conference on Machine Learning (ICML), 2023."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_52_1","DOI":"10.1109\/CVPR52729.2023.01471"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_53_1","DOI":"10.1109\/CVPR52688.2022.00928"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_54_1","DOI":"10.1109\/ICCV.2019.00474"},{"key":"e_1_3_2_1_55_1","first-page":"108","volume-title":"European Conference on Computer Vision","author":"Yang Yuhuan","year":"2024","unstructured":"Yuhuan Yang, Chaofan Ma, Jiangchao Yao, Zhun Zhong, Ya Zhang, and Yanfeng Wang. Remamber: Referring image segmentation with mamba twister. In European Conference on Computer Vision, pages 108--126. Springer, 2024."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_56_1","DOI":"10.1109\/CVPR52688.2022.01762"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_57_1","DOI":"10.1007\/978-3-030-58568-6_23"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_58_1","DOI":"10.1007\/978-3-031-20059-5_30"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_59_1","DOI":"10.1109\/ICCV.2019.00478"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_60_1","DOI":"10.1109\/CVPR.2018.00142"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_61_1","DOI":"10.1007\/978-3-319-46475-6_5"},{"key":"e_1_3_2_1_62_1","volume-title":"X 2-vlm: All-in-one pre-trained model for vision-language tasks","author":"Zeng Yan","year":"2023","unstructured":"Yan Zeng, Xinsong Zhang, Hang Li, Jiawei Wang, Jipeng Zhang, and Wangchunshu Zhou. X 2-vlm: All-in-one pre-trained model for vision-language tasks. IEEE Transactions on Pattern Analysis and Machine Intelligence, 2023."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_63_1","DOI":"10.1109\/TNNLS.2021.3090426"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_64_1","DOI":"10.1109\/ICCV48922.2021.00208"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_65_1","DOI":"10.1007\/978-3-031-19833-5_35"},{"key":"e_1_3_2_1_66_1","first-page":"4252","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Zhuang Bohan","year":"2018","unstructured":"Bohan Zhuang, Qi Wu, Chunhua Shen, Ian Reid, and Anton Van Den Hengel. Parallel attention: A unified framework for visual object discovery through dialogs and queries. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pages 4252--4261, 2018. graphy"}],"event":{"sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"acronym":"ICMR '25","name":"ICMR '25: International Conference on Multimedia Retrieval","location":"Chicago IL USA"},"container-title":["Proceedings of the 2025 International Conference on Multimedia Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3731715.3733274","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T04:06:52Z","timestamp":1755749212000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3731715.3733274"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,30]]},"references-count":66,"alternative-id":["10.1145\/3731715.3733274","10.1145\/3731715"],"URL":"https:\/\/doi.org\/10.1145\/3731715.3733274","relation":{},"subject":[],"published":{"date-parts":[[2025,6,30]]},"assertion":[{"value":"2025-06-30","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}