{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:24:39Z","timestamp":1750220679957,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":43,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,3,7]],"date-time":"2021-03-07T00:00:00Z","timestamp":1615075200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,3,7]]},"DOI":"10.1145\/3444685.3446270","type":"proceedings-article","created":{"date-parts":[[2021,5,4]],"date-time":"2021-05-04T04:48:41Z","timestamp":1620103721000},"page":"1-7","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Multi-level expression guided attention network for referring expression comprehension"],"prefix":"10.1145","author":[{"given":"Liang","family":"Peng","sequence":"first","affiliation":[{"name":"University of Electronic Science and Tenchnology of China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yang","family":"Yang","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Tenchnology of China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xing","family":"Xu","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Tenchnology of China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jingjing","family":"Li","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Tenchnology of China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaofeng","family":"Zhu","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Tenchnology of China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,5,3]]},"reference":[{"doi-asserted-by":"publisher","key":"e_1_3_2_1_1_1","DOI":"10.1109\/TCYB.2018.2831447"},{"key":"e_1_3_2_1_2_1","volume-title":"Dzmitry Bahdanau, and Yoshua Bengio.","author":"Cho Kyunghyun","year":"2014","unstructured":"Kyunghyun Cho , Bart Van Merri\u00ebnboer , Dzmitry Bahdanau, and Yoshua Bengio. 2014 . On the properties of neural machine translation: Encoder-decoder approaches. arXiv preprint arXiv:1409.1259 (2014). Kyunghyun Cho, Bart Van Merri\u00ebnboer, Dzmitry Bahdanau, and Yoshua Bengio. 2014. On the properties of neural machine translation: Encoder-decoder approaches. arXiv preprint arXiv:1409.1259 (2014)."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_3_1","DOI":"10.1109\/CVPR.2018.00808"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_4_1","DOI":"10.1162\/neco.1997.9.8.1735"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_5_1","DOI":"10.1109\/CVPR.2017.470"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_6_1","DOI":"10.1109\/CVPR.2016.493"},{"volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 770--778","author":"He","unstructured":"He K, Zhang X, Ren S, and Sun J . 2016. Deep residual learning for image recognition . In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 770--778 . He K, Zhang X, Ren S, and Sun J. 2016. Deep residual learning for image recognition. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 770--778.","key":"e_1_3_2_1_7_1"},{"unstructured":"Simonyan K and Zisserman A. 2014. Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014).  Simonyan K and Zisserman A. 2014. Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014).","key":"e_1_3_2_1_8_1"},{"key":"e_1_3_2_1_9_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980","author":"Kingma Diederik P","year":"2014","unstructured":"Diederik P Kingma and Jimmy Ba . 2014 . Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014). Diederik P Kingma and Jimmy Ba. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_10_1","DOI":"10.1007\/978-3-319-10602-1_48"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_11_1","DOI":"10.1109\/ICCV.2019.00477"},{"key":"e_1_3_2_1_12_1","volume-title":"Referring Expression Grounding by Marginalizing Scene Graph Likelihood. arXiv preprint arXiv:1906.03561","author":"Liu Daqing","year":"2019","unstructured":"Daqing Liu , Hanwang Zhang , Zheng-Jun Zha , and Fanglin Wang . 2019. Referring Expression Grounding by Marginalizing Scene Graph Likelihood. arXiv preprint arXiv:1906.03561 ( 2019 ). Daqing Liu, Hanwang Zhang, Zheng-Jun Zha, and Fanglin Wang. 2019. Referring Expression Grounding by Marginalizing Scene Graph Likelihood. arXiv preprint arXiv:1906.03561 (2019)."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_13_1","DOI":"10.1109\/CVPR.2017.333"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_14_1","DOI":"10.1016\/j.patcog.2017.02.034"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_15_1","DOI":"10.1109\/CVPR.2016.9"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_16_1","DOI":"10.1007\/978-3-319-46493-0_48"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_17_1","DOI":"10.1007\/s11042-018-6389-3"},{"key":"e_1_3_2_1_18_1","volume-title":"MRA-Net: Improving VQA via Multi-modal Relation Attention Network","author":"Peng Liang","year":"2020","unstructured":"Liang Peng , Yang Yang , Zheng Wang , Zi Huang , and Heng Tao Shen . 2020. MRA-Net: Improving VQA via Multi-modal Relation Attention Network . IEEE Transactions on Pattern Analysis and Machine Intelligence ( 2020 ). Liang Peng, Yang Yang, Zheng Wang, Zi Huang, and Heng Tao Shen. 2020. MRA-Net: Improving VQA via Multi-modal Relation Attention Network. IEEE Transactions on Pattern Analysis and Machine Intelligence (2020)."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_19_1","DOI":"10.1145\/3343031.3350925"},{"key":"e_1_3_2_1_20_1","volume-title":"Answer Again: Imporving VQA with Cascaded-Answering Model","author":"Peng Liang","year":"2020","unstructured":"Liang Peng , Yang Yang , Xiaopeng Zhang , Yanli Ji , Huimin Lu , and Heng Tao Shen . 2020 . Answer Again: Imporving VQA with Cascaded-Answering Model . IEEE Transactions on Knowledge and Data Engineering ( 2020), 1--12. Liang Peng, Yang Yang, Xiaopeng Zhang, Yanli Ji, Huimin Lu, and Heng Tao Shen. 2020. Answer Again: Imporving VQA with Cascaded-Answering Model. IEEE Transactions on Knowledge and Data Engineering (2020), 1--12."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_21_1","DOI":"10.3115\/v1\/D14-1162"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_22_1","DOI":"10.1109\/CVPR.2016.91"},{"unstructured":"Shaoqing Ren Kaiming He Ross Girshick and Jian Sun. 2015. Faster r-cnn: Towards real-time object detection with region proposal networks. In Advances in neural information processing systems. 91--99.  Shaoqing Ren Kaiming He Ross Girshick and Jian Sun. 2015. Faster r-cnn: Towards real-time object detection with region proposal networks. In Advances in neural information processing systems. 91--99.","key":"e_1_3_2_1_23_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_24_1","DOI":"10.1007\/978-3-319-46448-0_49"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_25_1","DOI":"10.18653\/v1\/W15-2812"},{"doi-asserted-by":"crossref","unstructured":"H. T. Shen L. Liu Y. Yang X. Xu Z. Huang F. Shen and R. Hong. 2020. Exploiting Subspace Relation in Semantic Labels for Cross-modal Hashing. IEEE Transactions on Knowledge and Data Engineering (2020).  H. T. Shen L. Liu Y. Yang X. Xu Z. Huang F. Shen and R. Hong. 2020. Exploiting Subspace Relation in Semantic Labels for Cross-modal Hashing. IEEE Transactions on Knowledge and Data Engineering (2020).","key":"e_1_3_2_1_26_1","DOI":"10.1109\/TKDE.2020.2970050"},{"unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez Lukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. In Advances in neural information processing systems. 5998--6008.  Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez Lukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. In Advances in neural information processing systems. 5998--6008.","key":"e_1_3_2_1_27_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_28_1","DOI":"10.1145\/3123266.3123326"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_29_1","DOI":"10.1109\/CVPR.2016.541"},{"volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 1960--1968","author":"Wang Peng","unstructured":"Peng Wang , Qi Wu , Jiewei Cao , Chunhua Shen , Lianli Gao , and Anton van den Hengel. 2019. Neighbourhood watch: Referring expression comprehension via language-guided graph attention networks . In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 1960--1968 . Peng Wang, Qi Wu, Jiewei Cao, Chunhua Shen, Lianli Gao, and Anton van den Hengel. 2019. Neighbourhood watch: Referring expression comprehension via language-guided graph attention networks. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 1960--1968.","key":"e_1_3_2_1_30_1"},{"key":"e_1_3_2_1_31_1","volume-title":"MUTATT: Visual-Textual Mutual Guidance for Referring Expression Comprehension. arXiv preprint arXiv:2003.08027","author":"Wang Shuai","year":"2020","unstructured":"Shuai Wang , Fan Lyu , Wei Feng , and Song Wang . 2020 . MUTATT: Visual-Textual Mutual Guidance for Referring Expression Comprehension. arXiv preprint arXiv:2003.08027 (2020). Shuai Wang, Fan Lyu, Wei Feng, and Song Wang. 2020. MUTATT: Visual-Textual Mutual Guidance for Referring Expression Comprehension. arXiv preprint arXiv:2003.08027 (2020)."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_32_1","DOI":"10.1016\/j.ipm.2019.102130"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_33_1","DOI":"10.1109\/CVPR42600.2020.01302"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_34_1","DOI":"10.1145\/3338533.3366552"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_35_1","DOI":"10.1109\/TIP.2017.2676345"},{"unstructured":"X. Xu T. Wang Y. Yang L. Zuo F. Shen and H. T. Shen. 2020. Cross-Modal Attention With Semantic Consistence for Image-Text Matching. IEEE Transactions on Neural Networks and Learning Systems (2020) 1--14.  X. Xu T. Wang Y. Yang L. Zuo F. Shen and H. T. Shen. 2020. Cross-Modal Attention With Semantic Consistence for Image-Text Matching. IEEE Transactions on Neural Networks and Learning Systems (2020) 1--14.","key":"e_1_3_2_1_36_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_37_1","DOI":"10.1109\/ICCV.2019.00474"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_38_1","DOI":"10.1109\/CVPR.2016.503"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_39_1","DOI":"10.1109\/CVPR.2018.00142"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_40_1","DOI":"10.1007\/978-3-319-46475-6_5"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_41_1","DOI":"10.1109\/CVPR.2017.375"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_42_1","DOI":"10.1109\/TIP.2018.2855415"},{"volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 4252--4261","author":"Zhuang Bohan","unstructured":"Bohan Zhuang , Qi Wu , Chunhua Shen , Ian Reid , and Anton van den Hengel. 2018. Parallel attention: A unified framework for visual object discovery through dialogs and queries . In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 4252--4261 . Bohan Zhuang, Qi Wu, Chunhua Shen, Ian Reid, and Anton van den Hengel. 2018. Parallel attention: A unified framework for visual object discovery through dialogs and queries. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 4252--4261.","key":"e_1_3_2_1_43_1"}],"event":{"sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"acronym":"MMAsia '20","name":"MMAsia '20: ACM Multimedia Asia","location":"Virtual Event Singapore"},"container-title":["Proceedings of the 2nd ACM International Conference on Multimedia in Asia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3444685.3446270","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3444685.3446270","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T22:03:19Z","timestamp":1750197799000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3444685.3446270"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,3,7]]},"references-count":43,"alternative-id":["10.1145\/3444685.3446270","10.1145\/3444685"],"URL":"https:\/\/doi.org\/10.1145\/3444685.3446270","relation":{},"subject":[],"published":{"date-parts":[[2021,3,7]]},"assertion":[{"value":"2021-05-03","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}