{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T01:22:00Z","timestamp":1784510520446,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":45,"publisher":"ACM","license":[{"start":{"date-parts":[[2020,10,12]],"date-time":"2020-10-12T00:00:00Z","timestamp":1602460800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Key R&D Program of Jiangxi Province","award":["No.20171ACH80022"],"award-info":[{"award-number":["No.20171ACH80022"]}]},{"name":"Natural Science Foundation of Guangdong Province in China","award":["No.2019B1515120049"],"award-info":[{"award-number":["No.2019B1515120049"]}]},{"name":"National Natural Science Foundation of China","award":["No.U1705262; No.61772443; No.61572410; No.61802324; No.617021"],"award-info":[{"award-number":["No.U1705262; No.61772443; No.61572410; No.61802324; No.617021"]}]},{"name":"National Key R&D Program","award":["No.2017YFC0113000; No.2016YFB1001503"],"award-info":[{"award-number":["No.2017YFC0113000; No.2016YFB1001503"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2020,10,12]]},"DOI":"10.1145\/3394171.3414006","type":"proceedings-article","created":{"date-parts":[[2020,10,12]],"date-time":"2020-10-12T12:26:53Z","timestamp":1602505613000},"page":"1274-1282","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":124,"title":["Cascade Grouped Attention Network for Referring Expression Segmentation"],"prefix":"10.1145","author":[{"given":"Gen","family":"Luo","sequence":"first","affiliation":[{"name":"Xiamen University, Xiamen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yiyi","family":"Zhou","sequence":"additional","affiliation":[{"name":"Xiamen University, Xiamen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rongrong","family":"Ji","sequence":"additional","affiliation":[{"name":"Xiamen University, Xiamen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaoshuai","family":"Sun","sequence":"additional","affiliation":[{"name":"Xiamen University, Xiamen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jinsong","family":"Su","sequence":"additional","affiliation":[{"name":"Xiamen University, Xiamen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chia-Wen","family":"Lin","sequence":"additional","affiliation":[{"name":"National Tsing Hua University, Taiwan, Taiwan Roc"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qi","family":"Tian","sequence":"additional","affiliation":[{"name":"Huawei Cloud BU, Huawei Technologies, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2020,10,12]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"crossref","unstructured":"Peter Anderson Xiaodong He Chris Buehler Damien Teney Mark Johnson Stephen Gould and Lei Zhang. 2018. Bottom-Up and Top-Down Attention for Image Captioning and Visual Question Answering. In CVPR.  Peter Anderson Xiaodong He Chris Buehler Damien Teney Mark Johnson Stephen Gould and Lei Zhang. 2018. Bottom-Up and Top-Down Attention for Image Captioning and Visual Question Answering. In CVPR.","DOI":"10.1109\/CVPR.2018.00636"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00755"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"crossref","unstructured":"Liangchieh Chen George Papandreou Iasonas Kokkinos Kevin P Murphy and Alan L Yuille. 2018. DeepLab: Semantic Image Segmentation with Deep Convolutional Nets Atrous Convolution and Fully Connected CRFs. In PAMI.  Liangchieh Chen George Papandreou Iasonas Kokkinos Kevin P Murphy and Alan L Yuille. 2018. DeepLab: Semantic Image Segmentation with Deep Convolutional Nets Atrous Convolution and Fully Connected CRFs. In PAMI.","DOI":"10.1109\/TPAMI.2017.2699184"},{"key":"e_1_3_2_2_4_1","unstructured":"Liang-Chieh Chen George Papandreou Iasonas Kokkinos Kevin Murphy and Alan L. Yuille. 2014. Semantic Image Segmentation with Deep Convolutional Nets and Fully Connected CRFs. In CVPR.  Liang-Chieh Chen George Papandreou Iasonas Kokkinos Kevin Murphy and Alan L. Yuille. 2014. Semantic Image Segmentation with Deep Convolutional Nets and Fully Connected CRFs. In CVPR."},{"key":"e_1_3_2_2_5_1","unstructured":"Junyoung Chung Caglar Gulcehre KyungHyun Cho and Yoshua Bengio. 2014. Empirical Evaluation of Gated Recurrent Neural Networks on Sequence Modeling. arXiv preprint arXiv:1412.3555 (2014).  Junyoung Chung Caglar Gulcehre KyungHyun Cho and Yoshua Bengio. 2014. Empirical Evaluation of Gated Recurrent Neural Networks on Sequence Modeling. arXiv preprint arXiv:1412.3555 (2014)."},{"key":"e_1_3_2_2_6_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805","author":"Devlin Jacob","year":"2018"},{"key":"e_1_3_2_2_7_1","unstructured":"John C Duchi Elad Hazan and Yoram Singer. 2011. Adaptive Subgradient Methods for Online Learning and Stochastic Optimization. In JMLR.  John C Duchi Elad Hazan and Yoram Singer. 2011. Adaptive Subgradient Methods for Online Learning and Stochastic Optimization. In JMLR."},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"crossref","unstructured":"Mark Everingham Luc Van Gool Christopher K I Williams John Winn and Andrew Zisserman. 2010. The Pascal Visual Object Classes (VOC) Challenge. In IJCV.  Mark Everingham Luc Van Gool Christopher K I Williams John Winn and Andrew Zisserman. 2010. The Pascal Visual Object Classes (VOC) Challenge. In IJCV.","DOI":"10.1007\/s11263-009-0275-4"},{"key":"e_1_3_2_2_9_1","unstructured":"Kaiming He Georgia Gkioxari Piotr Dollar and Ross B Girshick. 2017. Mask R-CNN. In ICCV.  Kaiming He Georgia Gkioxari Piotr Dollar and Ross B Girshick. 2017. Mask R-CNN. In ICCV."},{"key":"e_1_3_2_2_10_1","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep Residual Learning for Image Recognition. In CVPR.  Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep Residual Learning for Image Recognition. In CVPR."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"crossref","unstructured":"Sepp Hochreiter and Jurgen Schmidhuber. 1997. Long short-term memory. In Neural Computation.  Sepp Hochreiter and Jurgen Schmidhuber. 1997. Long short-term memory. In Neural Computation.","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"crossref","unstructured":"Sahar Kazemzadeh Vicente Ordonez Mark Matten and Tamara L Berg. 2014. ReferItGame: Referring to Objects in Photographs of Natural Scenes. In EMNLP.  Sahar Kazemzadeh Vicente Ordonez Mark Matten and Tamara L Berg. 2014. ReferItGame: Referring to Objects in Photographs of Natural Scenes. In EMNLP.","DOI":"10.3115\/v1\/D14-1086"},{"key":"e_1_3_2_2_13_1","unstructured":"Alex Krizhevsky Ilya Sutskever and Geoffrey E Hinton. 2012. Imagenet classification with deep convolutional neural networks. In NeurIPS.  Alex Krizhevsky Ilya Sutskever and Geoffrey E Hinton. 2012. Imagenet classification with deep convolutional neural networks. In NeurIPS."},{"key":"e_1_3_2_2_14_1","volume-title":"Dfanet: Deep feature aggregation for real-time semantic segmentation. In CVPR.","author":"Li Hanchao","year":"2019"},{"key":"e_1_3_2_2_15_1","unstructured":"Ruiyu Li Kaican Li Yichun Kuo Michelle Shu Xiaojuan Qi Xiaoyong Shen and Jiaya Jia. 2018. Referring Image Segmentation via Recurrent Refinement Networks. In CVPR.  Ruiyu Li Kaican Li Yichun Kuo Michelle Shu Xiaojuan Qi Xiaoyong Shen and Jiaya Jia. 2018. Referring Image Segmentation via Recurrent Refinement Networks. In CVPR."},{"key":"e_1_3_2_2_16_1","unstructured":"Guosheng Lin Anton Milan Chunhua Shen and Ian D Reid. 2017. RefineNet: Multi-path Refinement Networks for High-Resolution Semantic Segmentation. In CVPR.  Guosheng Lin Anton Milan Chunhua Shen and Ian D Reid. 2017. RefineNet: Multi-path Refinement Networks for High-Resolution Semantic Segmentation. In CVPR."},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00160"},{"key":"e_1_3_2_2_18_1","unstructured":"Tsungyi Lin Michael Maire Serge J Belongie James Hays Pietro Perona Deva Ramanan Piotr Dollar and C Lawrence Zitnick. 2014. Microsoft COCO: Common Objects in Context. In ECCV.  Tsungyi Lin Michael Maire Serge J Belongie James Hays Pietro Perona Deva Ramanan Piotr Dollar and C Lawrence Zitnick. 2014. Microsoft COCO: Common Objects in Context. In ECCV."},{"key":"e_1_3_2_2_19_1","unstructured":"Chenxi Liu Zhe Lin Xiaohui Shen Jimei Yang Xin Lu and Alan L Yuille. 2017. Recurrent Multimodal Interaction for Referring Image Segmentation. In ICCV.  Chenxi Liu Zhe Lin Xiaohui Shen Jimei Yang Xin Lu and Alan L Yuille. 2017. Recurrent Multimodal Interaction for Referring Image Segmentation. In ICCV."},{"key":"e_1_3_2_2_20_1","unstructured":"Daqing Liu Hanwang Zhang Feng Wu and Zheng-Jun Zha. 2019. Learning to assemble neural module tree networks for visual grounding. In ICCV.  Daqing Liu Hanwang Zhang Feng Wu and Zheng-Jun Zha. 2019. Learning to assemble neural module tree networks for visual grounding. In ICCV."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"crossref","unstructured":"Jonathan Long Evan Shelhamer and Trevor Darrell. 2015. Fully convolutional networks for semantic segmentation. In CVPR.  Jonathan Long Evan Shelhamer and Trevor Darrell. 2015. Fully convolutional networks for semantic segmentation. In CVPR.","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01005"},{"key":"e_1_3_2_2_23_1","unstructured":"Andrew L Maas Awni Y Hannun and Andrew Y Ng. 2013. Rectifier nonlinearities improve neural network acoustic models. In ICML.  Andrew L Maas Awni Y Hannun and Andrew Y Ng. 2013. Rectifier nonlinearities improve neural network acoustic models. In ICML."},{"key":"e_1_3_2_2_24_1","unstructured":"Junhua Mao Jonathan Huang Alexander Toshev Oana Maria Camburu Alan L Yuille and Kevin P Murphy. 2016. Generation and Comprehension of Unambiguous Object Descriptions. In CVPR.  Junhua Mao Jonathan Huang Alexander Toshev Oana Maria Camburu Alan L Yuille and Kevin P Murphy. 2016. Generation and Comprehension of Unambiguous Object Descriptions. In CVPR."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"crossref","unstructured":"Edgar A Margffoytuay Juan C Perez Emilio Botero and Pablo Andres Arbelaez. 2018. Dynamic Multimodal Instance Segmentation Guided by Natural Language Queries. In ECCV.  Edgar A Margffoytuay Juan C Perez Emilio Botero and Pablo Andres Arbelaez. 2018. Dynamic Multimodal Instance Segmentation Guided by Natural Language Queries. In ECCV.","DOI":"10.1007\/978-3-030-01252-6_39"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"crossref","unstructured":"Varun K Nagaraja Vlad I Morariu and Larry S Davis. 2016. Modeling Context Between Objects for Referring Expression Understanding. In ECCV.  Varun K Nagaraja Vlad I Morariu and Larry S Davis. 2016. Modeling Context Between Objects for Referring Expression Understanding. In ECCV.","DOI":"10.1007\/978-3-319-46493-0_48"},{"key":"e_1_3_2_2_27_1","unstructured":"Joseph Redmon and Ali Farhadi. 2018. YOLOv3: An Incremental Improvement. In arXiv preprint.  Joseph Redmon and Ali Farhadi. 2018. YOLOv3: An Incremental Improvement. In arXiv preprint."},{"key":"e_1_3_2_2_28_1","unstructured":"Trevor Darrell Ronghang Hu Marcus Rohrbach. 2016. Segmentation from Natural Language Expressions. In ECCV.  Trevor Darrell Ronghang Hu Marcus Rohrbach. 2016. Segmentation from Natural Language Expressions. In ECCV."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"crossref","unstructured":"Arka Sadhu Kan Chen and Ram Nevatia. 2019. Zero-Shot Grounding of Objects from Natural Language Queries. In ICCV.  Arka Sadhu Kan Chen and Ram Nevatia. 2019. Zero-Shot Grounding of Objects from Natural Language Queries. In ICCV.","DOI":"10.1109\/ICCV.2019.00479"},{"key":"e_1_3_2_2_30_1","unstructured":"Hengcan Shi Hongliang Li Fanman Meng and Qingbo Wu. 2018. Key-Word-Aware Network for Referring Expression Image Segmentation. In ECCV.  Hengcan Shi Hongliang Li Fanman Meng and Qingbo Wu. 2018. Key-Word-Aware Network for Referring Expression Image Segmentation. In ECCV."},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"crossref","unstructured":"Christian Szegedy Wei Liu Yangqing Jia Pierre Sermanet Scott Reed Dragomir Anguelov Dumitru Erhan Vincent Vanhoucke and Andrew Rabinovich. 2015. Going deeper with convolutions. In CVPR.  Christian Szegedy Wei Liu Yangqing Jia Pierre Sermanet Scott Reed Dragomir Anguelov Dumitru Erhan Vincent Vanhoucke and Andrew Rabinovich. 2015. Going deeper with convolutions. In CVPR.","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"crossref","unstructured":"Christian Szegedy Vincent Vanhoucke Sergey Ioffe Jon Shlens and Zbigniew Wojna. 2016. Rethinking the inception architecture for computer vision. In CVPR.  Christian Szegedy Vincent Vanhoucke Sergey Ioffe Jon Shlens and Zbigniew Wojna. 2016. Rethinking the inception architecture for computer vision. In CVPR.","DOI":"10.1109\/CVPR.2016.308"},{"key":"e_1_3_2_2_33_1","unstructured":"Yichuan Tang Nitish Srivastava and Ruslan R Salakhutdinov. 2014. Learning generative models with visual attention. In NeurIPS.  Yichuan Tang Nitish Srivastava and Ruslan R Salakhutdinov. 2014. Learning generative models with visual attention. In NeurIPS."},{"key":"e_1_3_2_2_34_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017a. Attention is all you need. In Advances in neural information processing systems. 5998--6008.  Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017a. Attention is all you need. In Advances in neural information processing systems. 5998--6008."},{"key":"e_1_3_2_2_35_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez Lukasz Kaiser and Illia Polosukhin. 2017b. Attention is All you Need. In NeurIPS.  Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez Lukasz Kaiser and Illia Polosukhin. 2017b. Attention is All you Need. In NeurIPS."},{"key":"e_1_3_2_2_36_1","unstructured":"Saining Xie Ross Girshick Piotr Doll\u00e1r Zhuowen Tu and Kaiming He. 2017. Aggregated residual transformations for deep neural networks. In CVPR.  Saining Xie Ross Girshick Piotr Doll\u00e1r Zhuowen Tu and Kaiming He. 2017. Aggregated residual transformations for deep neural networks. In CVPR."},{"key":"e_1_3_2_2_37_1","unstructured":"Huijuan Xu and Kate Saenko. 2016. Ask attend and answer: Exploring question-guided spatial attention for visual question answering. In ECCV.  Huijuan Xu and Kate Saenko. 2016. Ask attend and answer: Exploring question-guided spatial attention for visual question answering. In ECCV."},{"key":"e_1_3_2_2_38_1","unstructured":"Kelvin Xu Jimmy Ba Ryan Kiros Kyunghyun Cho Aaron C Courville Ruslan Salakhudinov Rich Zemel and Yoshua Bengio. 2015a. Show Attend and Tell: Neural Image Caption Generation with Visual Attention. In ICML.  Kelvin Xu Jimmy Ba Ryan Kiros Kyunghyun Cho Aaron C Courville Ruslan Salakhudinov Rich Zemel and Yoshua Bengio. 2015a. Show Attend and Tell: Neural Image Caption Generation with Visual Attention. In ICML."},{"key":"e_1_3_2_2_39_1","unstructured":"Kelvin Xu Jimmy Ba Ryan Kiros Kyunghyun Cho Aaron C Courville Ruslan Salakhudinov Rich Zemel and Yoshua Bengio. 2015b. Show Attend and Tell: Neural Image Caption Generation with Visual Attention. In ICML.  Kelvin Xu Jimmy Ba Ryan Kiros Kyunghyun Cho Aaron C Courville Ruslan Salakhudinov Rich Zemel and Yoshua Bengio. 2015b. Show Attend and Tell: Neural Image Caption Generation with Visual Attention. In ICML."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"crossref","unstructured":"Zhengyuan Yang Boqing Gong Liwei Wang Wenbing Huang Dong Yu and Jiebo Luo. 2019. A Fast and Accurate One-Stage Approach to Visual Grounding. In ICCV.  Zhengyuan Yang Boqing Gong Liwei Wang Wenbing Huang Dong Yu and Jiebo Luo. 2019. A Fast and Accurate One-Stage Approach to Visual Grounding. In ICCV.","DOI":"10.1109\/ICCV.2019.00478"},{"key":"e_1_3_2_2_41_1","volume":"201","author":"Yang Zichao","journal-title":"Alexander J Smola."},{"key":"e_1_3_2_2_42_1","unstructured":"Linwei Ye Mrigank Rochan Zhi Liu and Yang Wang. 2019. Cross-Modal Self-Attention Network for Referring Image Segmentation.. In CVPR.  Linwei Ye Mrigank Rochan Zhi Liu and Yang Wang. 2019. Cross-Modal Self-Attention Network for Referring Image Segmentation.. In CVPR."},{"key":"e_1_3_2_2_43_1","unstructured":"Licheng Yu Zhe Lin Xiaohui Shen Jimei Yang Xin Lu Mohit Bansal and Tamara L Berg. 2018. MAttNet: Modular Attention Network for Referring Expression Comprehension. In CVPR.  Licheng Yu Zhe Lin Xiaohui Shen Jimei Yang Xin Lu Mohit Bansal and Tamara L Berg. 2018. MAttNet: Modular Attention Network for Referring Expression Comprehension. In CVPR."},{"key":"e_1_3_2_2_44_1","unstructured":"Hengshuang Zhao Jianping Shi Xiaojuan Qi Xiaogang Wang and Jiaya Jia. 2017. Pyramid Scene Parsing Network. In CVPR.  Hengshuang Zhao Jianping Shi Xiaojuan Qi Xiaogang Wang and Jiaya Jia. 2017. Pyramid Scene Parsing Network. In CVPR."},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"crossref","unstructured":"Yiyi Zhou Rongrong Ji Jinsong Su Xiaoshuai Sun and Weiqiu Chen. 2019. Dynamic Capsule Attention for Visual Question Answering. (2019).  Yiyi Zhou Rongrong Ji Jinsong Su Xiaoshuai Sun and Weiqiu Chen. 2019. Dynamic Capsule Attention for Visual Question Answering. (2019).","DOI":"10.1609\/aaai.v33i01.33019324"}],"event":{"name":"MM '20: The 28th ACM International Conference on Multimedia","location":"Seattle WA USA","acronym":"MM '20","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 28th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3394171.3414006","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3394171.3414006","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T21:32:07Z","timestamp":1750195927000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3394171.3414006"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,10,12]]},"references-count":45,"alternative-id":["10.1145\/3394171.3414006","10.1145\/3394171"],"URL":"https:\/\/doi.org\/10.1145\/3394171.3414006","relation":{},"subject":[],"published":{"date-parts":[[2020,10,12]]},"assertion":[{"value":"2020-10-12","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}