{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,2]],"date-time":"2025-12-02T03:34:42Z","timestamp":1764646482092,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":37,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Key R & D projects of Shaanxi Province, China","award":["2023-GHZD-02"],"award-info":[{"award-number":["2023-GHZD-02"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62271400"],"award-info":[{"award-number":["62271400"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Research Foundation, Singapore","award":["NRFF award NRF-NRFF13-2021-0008"],"award-info":[{"award-number":["NRFF award NRF-NRFF13-2021-0008"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3612493","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:30Z","timestamp":1698391650000},"page":"2507-2515","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Transformer-based Open-world Instance Segmentation with Cross-task Consistency Regularization"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8636-3993","authenticated-orcid":false,"given":"Xizhe","family":"Xue","sequence":"first","affiliation":[{"name":"Northwestern Polytechnical University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4967-022X","authenticated-orcid":false,"given":"Dongdong","family":"Yu","sequence":"additional","affiliation":[{"name":"ByteDance Inc., Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3584-795X","authenticated-orcid":false,"given":"Lingqiao","family":"Liu","sequence":"additional","affiliation":[{"name":"The University of Adelaide, Adelaide, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0139-0536","authenticated-orcid":false,"given":"Yu","family":"Liu","sequence":"additional","affiliation":[{"name":"ByteDance Inc., Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8003-2616","authenticated-orcid":false,"given":"Satoshi","family":"Tsutsui","sequence":"additional","affiliation":[{"name":"Nanyang Technological University, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7370-1754","authenticated-orcid":false,"given":"Ying","family":"Li","sequence":"additional","affiliation":[{"name":"Northwestern Polytechnical University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0349-9367","authenticated-orcid":false,"given":"Zehuan","family":"Yuan","sequence":"additional","affiliation":[{"name":"ByteDance Inc., Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3353-1444","authenticated-orcid":false,"given":"Ping","family":"Song","sequence":"additional","affiliation":[{"name":"ByteDance Inc., Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7681-2166","authenticated-orcid":false,"given":"Mike Zheng","family":"Shou","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00925"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00860"},{"key":"e_1_3_2_2_4_1","volume-title":"Masked-attention Mask Transformer for Universal Image Segmentation. In 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 1280--1289","author":"Cheng Bowen","year":"2022","unstructured":"Bowen Cheng, Ishan Misra, Alexander G. Schwing, Alexander Kirillov, and Rohit Girdhar. 2022. Masked-attention Mask Transformer for Universal Image Segmentation. In 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 1280--1289."},{"key":"e_1_3_2_2_5_1","first-page":"17864","article-title":"Per-pixel classification is not all you need for semantic segmentation","volume":"34","author":"Cheng Bowen","year":"2021","unstructured":"Bowen Cheng, Alex Schwing, and Alexander Kirillov. 2021. Per-pixel classification is not all you need for semantic segmentation. Advances in Neural Information Processing Systems, Vol. 34 (2021), 17864--17875.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"crossref","unstructured":"Tianheng Cheng Xinggang Wang Lichao Huang and Wenyu Liu. 2020. Boundary-preserving mask r-cnn. (2020) 660--676.","DOI":"10.1007\/978-3-030-58568-6_39"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.350"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.343"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_2_10_1","volume-title":"Improved regularization of convolutional neural networks with cutout. arXiv preprint arXiv:1708.04552","author":"DeVries Terrance","year":"2017","unstructured":"Terrance DeVries and Graham W Taylor. 2017. Improved regularization of convolutional neural networks with cutout. arXiv preprint arXiv:1708.04552 (2017)."},{"key":"e_1_3_2_2_11_1","first-page":"21898","article-title":"Solq: Segmenting objects by learning queries","volume":"34","author":"Dong Bin","year":"2021","unstructured":"Bin Dong, Fangao Zeng, Tiancai Wang, Xiangyu Zhang, and Yichen Wei. 2021. Solq: Segmenting objects by learning queries. Advances in Neural Information Processing Systems, Vol. 34 (2021), 21898--21909.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_12_1","volume-title":"1st Place Solution for the UVO Challenge on Video-based Open-World Segmentation","author":"Du Yuming","year":"2021","unstructured":"Yuming Du, Wen Guo, Yang Xiao, and Vincent Lepetit. 2021. 1st Place Solution for the UVO Challenge on Video-based Open-World Segmentation 2021. arXiv preprint arXiv:2110.11661 (2021)."},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00683"},{"key":"e_1_3_2_2_14_1","volume-title":"Yolox: Exceeding yolo series in","author":"Ge Zheng","year":"2021","unstructured":"Zheng Ge, Songtao Liu, Feng Wang, Zeming Li, and Jian Sun. 2021. Yolox: Exceeding yolo series in 2021. arXiv preprint arXiv:2107.08430 (2021)."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.322"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"crossref","unstructured":"Ronghang Hu Piotr Doll\u00e1r Kaiming He Trevor Darrell and Ross Girshick. 2018. Learning to segment every thing. (2018) 4233--4241.","DOI":"10.1109\/CVPR.2018.00445"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"crossref","unstructured":"Lei Ke Martin Danelljan Xia Li Yu-Wing Tai Chi-Keung Tang and Fisher Yu. 2022. Mask Transfiner for High-Quality Instance Segmentation. (2022) 4412--4421.","DOI":"10.1109\/CVPR52688.2022.00437"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"crossref","unstructured":"Lei Ke Yu-Wing Tai and Chi-Keung Tang. 2021. Deep Occlusion-Aware Instance Segmentation with Overlapping BiLayers. (2021) 4019--4028.","DOI":"10.1109\/CVPR46437.2021.00401"},{"key":"e_1_3_2_2_20_1","volume-title":"arXiv:2304.02643","author":"Kirillov Alexander","year":"2023","unstructured":"Alexander Kirillov, Eric Mintun, Nikhila Ravi, Hanzi Mao, Chloe Rolland, Laura Gustafson, Tete Xiao, Spencer Whitehead, Alexander C. Berg, Wan-Yen Lo, Piotr Doll\u00e1r, and Ross Girshick. 2023. Segment Anything. arXiv:2304.02643 (2023)."},{"key":"e_1_3_2_2_21_1","volume-title":"Pointrend: Image segmentation as rendering.","author":"Kirillov Alexander","year":"2020","unstructured":"Alexander Kirillov, Yuxin Wu, Kaiming He, and Ross Girshick. 2020. Pointrend: Image segmentation as rendering. (2020), 9799--9808."},{"key":"e_1_3_2_2_22_1","volume-title":"Shapemask: Learning to segment novel objects by refining shape priors.","author":"Kuo Weicheng","year":"2019","unstructured":"Weicheng Kuo, Anelia Angelova, Jitendra Malik, and Tsung-Yi Lin. 2019. Shapemask: Learning to segment novel objects by refining shape priors. (2019), 9207--9216."},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01831"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00859"},{"key":"e_1_3_2_2_27_1","volume-title":"V-net: Fully convolutional neural networks for volumetric medical image segmentation. In 2016 fourth international conference on 3D vision (3DV)","author":"Milletari Fausto","year":"2016","unstructured":"Fausto Milletari, Nassir Navab, and Seyed-Ahmad Ahmadi. 2016. V-net: Fully convolutional neural networks for volumetric medical image segmentation. In 2016 fourth international conference on 3D vision (3DV). IEEE, 565--571."},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.534"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20053-3_16"},{"key":"e_1_3_2_2_30_1","volume-title":"Cascade rpn: Delving into high-quality region proposal network with adaptive convolution. Advances in neural information processing systems","author":"Vu Thang","year":"2019","unstructured":"Thang Vu, Hyunjun Jang, Trung X Pham, and Chang Yoo. 2019. Cascade rpn: Delving into high-quality region proposal network with adaptive convolution. Advances in neural information processing systems, Vol. 32 (2019)."},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"crossref","unstructured":"Weiyao Wang Matt Feiszli Heng Wang Jitendra Malik and Du Tran. 2022. Open-World Instance Segmentation: Exploiting Pseudo Ground Truth From Learned Pairwise Affinity. (2022) 4422--4432.","DOI":"10.1109\/CVPR52688.2022.00438"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01060"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58523-5_38"},{"key":"e_1_3_2_2_34_1","volume-title":"SOLOv2: Dynamic and Fast Instance Segmentation. Advances in Neural information processing systems","author":"Wang Xinlong","year":"2020","unstructured":"Xinlong Wang, Rufeng Zhang, Tao Kong, Lei Li, and Chunhua Shen. 2020b. SOLOv2: Dynamic and Fast Instance Segmentation. Advances in Neural information processing systems, Vol. 33 (2020), 17721--17732."},{"key":"e_1_3_2_2_35_1","volume-title":"https:\/\/github.com\/facebookresearch\/detectron2","author":"Wu Yuxin","year":"2019","unstructured":"Yuxin Wu, Alexander Kirillov, Francisco Massa, Wan-Yen Lo, and Ross. Girshick. 2019. Detectron2. (2019). https:\/\/github.com\/facebookresearch\/detectron2"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01121"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1093\/nsr\/nwx105"}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Ottawa ON Canada","acronym":"MM '23"},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612493","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3612493","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:01:11Z","timestamp":1755820871000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612493"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":37,"alternative-id":["10.1145\/3581783.3612493","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3612493","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}