{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,27]],"date-time":"2026-03-27T16:13:13Z","timestamp":1774627993383,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":37,"publisher":"ACM","funder":[{"name":"Hong Kong RGC General Research Fund","award":["152169\/22E, 152228\/23E, 162161\/24E"],"award-info":[{"award-number":["152169\/22E, 152228\/23E, 162161\/24E"]}]},{"name":"Research Impact Fund","award":["No. R5060-19, No. R5011-23"],"award-info":[{"award-number":["No. R5060-19, No. R5011-23"]}]},{"name":"Collaborative Research Fund","award":["No. C1042-23GF"],"award-info":[{"award-number":["No. C1042-23GF"]}]},{"name":"NSFC\/RGC Collaborative Research Scheme","award":["Grant No. 62461160332 \\& CRS\\_HKUST602\/24"],"award-info":[{"award-number":["Grant No. 62461160332 \\& CRS\\_HKUST602\/24"]}]},{"name":"Areas of Excellence Scheme","award":["AoE\/E-601\/22-R"],"award-info":[{"award-number":["AoE\/E-601\/22-R"]}]},{"name":"InnoHK","award":["HKGAI"],"award-info":[{"award-number":["HKGAI"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755019","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T05:47:42Z","timestamp":1761371262000},"page":"238-246","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["BoxSeg: Quality-Aware and Peer-Assisted Learning for Box-supervised Instance Segmentation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-1873-9856","authenticated-orcid":false,"given":"Jinxiang","family":"Lai","sequence":"first","affiliation":[{"name":"CSE, The Hong Kong University of Science and Technology, Hong Kong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3732-6456","authenticated-orcid":false,"given":"Wenlong","family":"Wu","sequence":"additional","affiliation":[{"name":"Tencent YouTu Lab, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1745-4768","authenticated-orcid":false,"given":"Jiawei","family":"Zhan","sequence":"additional","affiliation":[{"name":"Tencent Youtu Lab, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0242-6481","authenticated-orcid":false,"given":"Jian","family":"Li","sequence":"additional","affiliation":[{"name":"Tencent YouTu Lab, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2572-8156","authenticated-orcid":false,"given":"Bin-Bin","family":"Gao","sequence":"additional","affiliation":[{"name":"Tencent YouTu Lab, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6985-8238","authenticated-orcid":false,"given":"Jun","family":"Liu","sequence":"additional","affiliation":[{"name":"Tencent YouTu Lab, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8073-2118","authenticated-orcid":false,"given":"Jie","family":"Zhang","sequence":"additional","affiliation":[{"name":"CSE, The Hong Kong University of Science and Technology, Hong Kong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9831-2202","authenticated-orcid":false,"given":"Song","family":"Guo","sequence":"additional","affiliation":[{"name":"CSE, The Hong Kong University of Science and Technology, Hong Kong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Jiwoon Ahn Sunghyun Cho and Suha Kwak. 2019. Weakly Supervised Learning of Instance Segmentation With Inter-Pixel Relations. In CVPR."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"crossref","unstructured":"Aditya Arun C. V. Jawahar and M. Pawan Kumar. 2020. Weakly Supervised Instance Segmentation by Learning Annotation Consistent Instances. In ECCV.","DOI":"10.1007\/978-3-030-58604-1_16"},{"key":"e_1_3_2_1_3_1","volume-title":"YOLACT: Real-Time Instance Segmentation. In ICCV.","author":"Bolya Daniel","year":"2019","unstructured":"Daniel Bolya, Chong Zhou, Fanyi Xiao, and Yong Jae Lee. 2019. YOLACT: Real-Time Instance Segmentation. In ICCV."},{"key":"e_1_3_2_1_4_1","volume-title":"Cascade R-CNN: High Quality Object Detection and Instance Segmentation","author":"Cai Zhaowei","year":"2021","unstructured":"Zhaowei Cai and Nuno Vasconcelos. 2021. Cascade R-CNN: High Quality Object Detection and Instance Segmentation. IEEE Trans. Pattern Anal. Mach. Intell. (2021), 1483--1498."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"Nicolas Carion Francisco Massa Gabriel Synnaeve Nicolas Usunier Alexander Kirillov and Sergey Zagoruyko. 2020. End-to-End Object Detection with Transformers. In ECCV.","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"e_1_3_2_1_6_1","volume-title":"Dynamic Convolution: Attention Over Convolution Kernels. In CVPR.","author":"Chen Yinpeng","year":"2020","unstructured":"Yinpeng Chen, Xiyang Dai, Mengchen Liu, Dongdong Chen, Lu Yuan, and Zicheng Liu. 2020. Dynamic Convolution: Attention Over Convolution Kernels. In CVPR."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Bowen Cheng Ishan Misra Alexander G. Schwing Alexander Kirillov and Rohit Girdhar. 2021. Masked-attention Mask Transformer for Universal Image Segmentation. In CVPR.","DOI":"10.1109\/CVPR52688.2022.00135"},{"key":"e_1_3_2_1_8_1","unstructured":"Bowen Cheng Alexander G. Schwing and Alexander Kirillov. 2021. Per-Pixel Classification is Not All You Need for Semantic Segmentation. In NeurIPS."},{"key":"e_1_3_2_1_9_1","volume-title":"Boxteacher: Exploring high-quality pseudo labels for weakly supervised instance segmentation. In CVPR.","author":"Cheng Tianheng","year":"2023","unstructured":"Tianheng Cheng, Xinggang Wang, Shaoyu Chen, Qian Zhang, and Wenyu Liu. 2023. Boxteacher: Exploring high-quality pseudo labels for weakly supervised instance segmentation. In CVPR."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"crossref","unstructured":"Tianheng Cheng Xinggang Wang Lichao Huang and Wenyu Liu. 2020. Boundary-preserving Mask R-CNN. In ECCV.","DOI":"10.1007\/978-3-030-58568-6_39"},{"key":"e_1_3_2_1_11_1","volume-title":"Christopher K. I. Williams, John M. Winn, and Andrew Zisserman.","author":"Everingham Mark","year":"2010","unstructured":"Mark Everingham, Luc Van Gool, Christopher K. I. Williams, John M. Winn, and Andrew Zisserman. 2010. The Pascal Visual Object Classes (VOC) Challenge. Int. J. Comput. Vis. (2010)."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"crossref","unstructured":"Yuxin Fang Shusheng Yang Xinggang Wang Yu Li Chen Fang Ying Shan Bin Feng and Wenyu Liu. 2021. Instances as Queries. In ICCV.","DOI":"10.1109\/ICCV48922.2021.00683"},{"key":"e_1_3_2_1_13_1","volume-title":"Girshick","author":"He Kaiming","year":"2017","unstructured":"Kaiming He, Georgia Gkioxari, Piotr Doll\u00e1r, and Ross B. Girshick. 2017. Mask R-CNN. In ICCV."},{"key":"e_1_3_2_1_14_1","unstructured":"Cheng-Chun Hsu Kuang-Jui Hsu Chung-Chi Tsai Yen-Yu Lin and Yung-Yu Chuang. 2019. Weakly Supervised Instance Segmentation using the Bounding Box Tightness Prior. In NeurIPS. 6582--6593."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"crossref","unstructured":"Zhaojin Huang Lichao Huang Yongchao Gong Chang Huang and Xinggang Wang. 2019. Mask Scoring R-CNN. In CVPR.","DOI":"10.1109\/CVPR.2019.00657"},{"key":"e_1_3_2_1_16_1","volume-title":"Girshick","author":"Kirillov Alexander","year":"2020","unstructured":"Alexander Kirillov, YuxinWu, Kaiming He, and Ross B. Girshick. 2020. PointRend: Image Segmentation As Rendering. In CVPR."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","unstructured":"Shiyi Lan Zhiding Yu Christopher B. Choy Subhashree Radhakrishnan Guilin Liu Yuke Zhu Larry S. Davis and Anima Anandkumar. 2021. DiscoBox: Weakly Supervised Instance Segmentation and Semantic Correspondence from Box Supervision. In ICCV.","DOI":"10.1109\/ICCV48922.2021.00339"},{"key":"e_1_3_2_1_18_1","volume-title":"BBAM: Bounding Box Attribution Map for Weakly Supervised Semantic and Instance Segmentation. In CVPR.","author":"Lee Jungbeom","year":"2021","unstructured":"Jungbeom Lee, Jihun Yi, Chaehun Shin, and Sungroh Yoon. 2021. BBAM: Bounding Box Attribution Map for Weakly Supervised Semantic and Instance Segmentation. In CVPR."},{"key":"e_1_3_2_1_19_1","volume-title":"SIM: Semantic-aware Instance Mask Generation for Box-Supervised Instance Segmentation. In CVPR.","author":"Li Ruihuang","year":"2023","unstructured":"Ruihuang Li, Chenhang He, Yabin Zhang, Shuai Li, Liyi Chen, and Lei Zhang. 2023. SIM: Semantic-aware Instance Mask Generation for Box-Supervised Instance Segmentation. In CVPR."},{"key":"e_1_3_2_1_20_1","unstructured":"Wentong Li Wenyu Liu Jianke Zhu Miaomiao Cui Xiansheng Hua and Lei Zhang. 2022. Box-supervised Instance Segmentation with Level Set Evolution. In ECCV."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3363054"},{"key":"e_1_3_2_1_22_1","unstructured":"Tsung-Yi Lin Michael Maire Serge J. Belongie James Hays Pietro Perona Deva Ramanan Piotr Doll\u00e1r and C. Lawrence Zitnick. 2014. Microsoft COCO: Common Objects in Context. In ECCV."},{"key":"e_1_3_2_1_23_1","volume-title":"Berg","author":"Liu Wei","year":"2016","unstructured":"Wei Liu, Dragomir Anguelov, Dumitru Erhan, Christian Szegedy, Scott E. Reed, Cheng-Yang Fu, and Alexander C. Berg. 2016. SSD: Single Shot MultiBox Detector. In ECCV."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"crossref","unstructured":"Fausto Milletari Nassir Navab and Seyed-Ahmad Ahmadi. 2016. V-Net: Fully Convolutional Neural Networks for Volumetric Medical Image Segmentation. In 3DV.","DOI":"10.1109\/3DV.2016.79"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"crossref","unstructured":"Zhi Tian Chunhua Shen and Hao Chen. 2020. Conditional Convolutions for Instance Segmentation. In ECCV.","DOI":"10.1007\/978-3-030-58452-8_17"},{"key":"e_1_3_2_1_26_1","volume-title":"FCOS: Fully Convolutional One-Stage Object Detection. In ICCV.","author":"Tian Zhi","year":"2019","unstructured":"Zhi Tian, Chunhua Shen, Hao Chen, and Tong He. 2019. FCOS: Fully Convolutional One-Stage Object Detection. In ICCV."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","unstructured":"Zhi Tian Chunhua Shen Xinlong Wang and Hao Chen. 2021. BoxInst: High-Performance Instance Segmentation With Box Annotations. In CVPR.","DOI":"10.1109\/CVPR46437.2021.00540"},{"key":"e_1_3_2_1_28_1","volume-title":"Peer-assisted learning","author":"Topping Keith","unstructured":"Keith Topping and Stewart Ehly. 1998. Peer-assisted learning. Routledge."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"crossref","unstructured":"Xinggang Wang Jiapei Feng Bin Hu Qi Ding Longjin Ran Xiaoxin Chen and Wenyu Liu. 2021. Weakly-Supervised Instance Segmentation via Class-Agnostic Learning With Salient Images. In CVPR.","DOI":"10.1109\/CVPR46437.2021.01009"},{"key":"e_1_3_2_1_30_1","volume-title":"SOLO: Segmenting Objects by Locations. In ECCV.","author":"Wang Xinlong","year":"2020","unstructured":"Xinlong Wang, Tao Kong, Chunhua Shen, Yuning Jiang, and Lei Li. 2020. SOLO: Segmenting Objects by Locations. In ECCV."},{"key":"e_1_3_2_1_31_1","unstructured":"Xinlong Wang Rufeng Zhang Tao Kong Lei Li and Chunhua Shen. 2020. SOLOv2: Dynamic and Fast Instance Segmentation. In NeurIPS."},{"key":"e_1_3_2_1_32_1","unstructured":"Yuxin Wu Alexander Kirillov Francisco Massa Wan-Yen Lo and Ross Girshick. 2019. Detectron2. https:\/\/github.com\/facebookresearch\/detectron2."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"crossref","unstructured":"Enze Xie Peize Sun Xiaoge Song WenhaiWang Xuebo Liu Ding Liang Chunhua Shen and Ping Luo. 2020. PolarMask: Single Shot Instance Segmentation With Polar Representation. In CVPR.","DOI":"10.1109\/CVPR42600.2020.01221"},{"key":"e_1_3_2_1_34_1","volume-title":"Varifocalnet: An iou-aware dense object detector. In CVPR.","author":"Zhang Haoyang","year":"2021","unstructured":"Haoyang Zhang, Ying Wang, Feras Dayoub, and Niko Sunderhauf. 2021. Varifocalnet: An iou-aware dense object detector. In CVPR."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"crossref","unstructured":"Rufeng Zhang Zhi Tian Chunhua Shen Mingyu You and Youliang Yan. 2020. Mask Encoding for Single Shot Instance Segmentation. In CVPR.","DOI":"10.1109\/CVPR42600.2020.01024"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"crossref","unstructured":"Yanzhao Zhou Yi Zhu Qixiang Ye Qiang Qiu and Jianbin Jiao. 2018. Weakly Supervised Instance Segmentation Using Class Peak Response. In CVPR.","DOI":"10.1109\/CVPR.2018.00399"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"crossref","unstructured":"Yi Zhu Yanzhao Zhou Huijuan Xu Qixiang Ye David S. Doermann and Jianbin Jiao. 2019. Learning Instance Activation Maps for Weakly Supervised Instance Segmentation. In CVPR.","DOI":"10.1109\/CVPR.2019.00323"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755019","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:20:50Z","timestamp":1765308050000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755019"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":37,"alternative-id":["10.1145\/3746027.3755019","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755019","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}