{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:20:34Z","timestamp":1750220434765,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":51,"publisher":"ACM","license":[{"start":{"date-parts":[[2020,10,12]],"date-time":"2020-10-12T00:00:00Z","timestamp":1602460800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Natural Science Foundation of China","award":["61532018,U1936203"],"award-info":[{"award-number":["61532018,U1936203"]}]},{"name":"National Postdoctoral Program for Innovative Talents","award":["BX20200338"],"award-info":[{"award-number":["BX20200338"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2020,10,12]]},"DOI":"10.1145\/3394171.3413567","type":"proceedings-article","created":{"date-parts":[[2020,10,12]],"date-time":"2020-10-12T12:27:38Z","timestamp":1602505658000},"page":"2581-2589","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Expressional Region Retrieval"],"prefix":"10.1145","author":[{"given":"Xiaoqian","family":"Guo","sequence":"first","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiangyang","family":"Li","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuqiang","family":"Jiang","sequence":"additional","affiliation":[{"name":"ICT, China Academy of Science, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2020,10,12]]},"reference":[{"volume-title":"Language Features Matter: Effective Language Representations for Vision-Language Tasks. In IEEE International Conference on Computer Vision. 7473--7482","author":"Burns A.","key":"e_1_3_2_2_1_1"},{"volume":"3","volume-title":"Hawaii International Conference on System Sciences","author":"Chua T.","key":"e_1_3_2_2_2_1"},{"volume-title":"IEEE Conference on Computer Vision and Pattern Recognition. 7746--7755","author":"Deng C.","key":"e_1_3_2_2_3_1"},{"volume-title":"Proceedings of the EACL 2014 Workshop on Statistical Machine Translation. 376--380","author":"Denkowski M.","key":"e_1_3_2_2_4_1"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2010.266"},{"volume-title":"Proceedings of the 26th International Conference on Neural Information Processing Systems. 2121--2129","author":"Frome A.","key":"e_1_3_2_2_6_1"},{"volume-title":"Deep Image Retrieval: Learning Global Representations for Image Search. In European Conference on Computer Vision. 241--257","author":"Gordo A.","key":"e_1_3_2_2_7_1"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"crossref","unstructured":"S. Guadarrama E. Rodner K. Saenko N. Zhang R. Farrell J. Donahue and T. Darrell. 2014. Open-vocabulary Object Retrieval. In Robotics Science and Systems.  S. Guadarrama E. Rodner K. Saenko N. Zhang R. Farrell J. Donahue and T. Darrell. 2014. Open-vocabulary Object Retrieval. In Robotics Science and Systems.","DOI":"10.15607\/RSS.2014.X.041"},{"volume-title":"Deep Residual Learning for Image Recognition. In IEEE Conference on Computer Vision and Pattern Recognition. 770--778","author":"He K.","key":"e_1_3_2_2_9_1"},{"key":"e_1_3_2_2_10_1","volume-title":"Learning Distance Metrics with Contextual Constraints for Image Retrieval. In IEEE Conference on Computer Vision and Pattern Recognition","volume":"2","author":"Hoi S. C. H.","year":"2072"},{"volume-title":"Modeling Relationships in Referential Expressions with Compositional Modular Networks. In IEEE Conference on Computer Vision and Pattern Recognition. 4418--4427","author":"Hu R.","key":"e_1_3_2_2_11_1"},{"volume-title":"Natural Language Object Retrieval. In IEEE Conference on Computer Vision and Pattern Recognition. 4555--4564","author":"Hu R.","key":"e_1_3_2_2_12_1"},{"volume-title":"Proceedings of the 16th ACM International Conference on Multimedia. 141--150","author":"Hua X.","key":"e_1_3_2_2_13_1"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2019.2891124"},{"volume-title":"DenseCap: Fully Convolutional Localization Networks for Dense Captioning. In IEEE Conference on Computer Vision and Pattern Recognition. 4565--4574","author":"Johnson J.","key":"e_1_3_2_2_15_1"},{"volume-title":"IEEE Conference on Computer Vision and Pattern Recognition. 3668--3678","author":"Johnson J.","key":"e_1_3_2_2_16_1"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-016-0981-7"},{"volume-title":"Supervised Ranking Hash for Semantic Similarity Search. In IEEE International Symposium on Multimedia. 551--558","author":"Li K.","key":"e_1_3_2_2_18_1"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/MMUL.2013.15"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2018.2811621"},{"key":"e_1_3_2_2_21_1","volume-title":"ACM Comput. Surv.","volume":"49","author":"Li X.","year":"2016"},{"volume-title":"Learning Deep Representations for Ground-to-Aerial Geolocalization. In IEEE Conference on Computer Vision and Pattern Recognition. 5007--5015","author":"Lin T.","key":"e_1_3_2_2_22_1"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2012.06.001"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1023\/B:VISI.0000029664.99615.94"},{"volume-title":"Comprehension-Guided Referring Expressions. In IEEE Conference on Computer Vision and Pattern Recognition. 3125--3134","author":"Luo R.","key":"e_1_3_2_2_25_1"},{"volume-title":"Generation and Comprehension of Unambiguous Object Descriptions. In IEEE Conference on Computer Vision and Pattern Recognition. 11--20","author":"Mao J.","key":"e_1_3_2_2_26_1"},{"key":"e_1_3_2_2_27_1","unstructured":"A. Miech I. Laptev and J. Sivic. 2017. Learnable pooling with Context Gating for video classification. CoRR Vol. abs\/1706.06905 (2017).  A. Miech I. Laptev and J. Sivic. 2017. Learnable pooling with Context Gating for video classification. CoRR Vol. abs\/1706.06905 (2017)."},{"volume-title":"Linguistic Regularities in Continuous Space Word Representations. In Human Language Technologies: Conference of the North American Chapter of the Association of Computational Linguistics. 746--751","author":"Mikolov T.","key":"e_1_3_2_2_28_1"},{"volume-title":"IEEE Conference on Computer Vision and Pattern Recognition Workshops. 53--61","author":"Ng J. Y.","key":"e_1_3_2_2_29_1"},{"volume-title":"Large-Scale Image Retrieval with Attentive Deep Local Features. In IEEE International Conference on Computer Vision. 3476--3485","author":"Noh H.","key":"e_1_3_2_2_30_1"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0840-y"},{"volume-title":"Proceedings of the 40th Annual Meeting on Association for Computational Linguistics. 311--318","author":"Papineni K.","key":"e_1_3_2_2_32_1"},{"volume-title":"Phrase Localization and Visual Relationship Detection with Comprehensive Image-Language Cues. In IEEE International Conference on Computer Vision. 1946--1955","author":"Plummer B. A.","key":"e_1_3_2_2_33_1"},{"key":"e_1_3_2_2_34_1","first-page":"251","article-title":"Visual Instance Retrieval with Deep Convolutional Networks","volume":"4","author":"Razavian A. S.","year":"2016","journal-title":"ITE Trans. MTA"},{"volume-title":"Proceedings of the 28th International Conference on Neural Information Processing Systems. 91--99","author":"Ren S.","key":"e_1_3_2_2_35_1"},{"volume-title":"Proceedings of the 20th ACM International Conference on Multimedia. ACM, 965--968","author":"Revaud J.","key":"e_1_3_2_2_36_1"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"volume-title":"Proceedings of the 2016 ACM on International Conference on Multimedia Retrieval. 243--247","author":"Chowdhury S. N.","key":"e_1_3_2_2_38_1"},{"volume-title":"IEEE Conference on Computer Vision and Pattern Recognition. 801--808","author":"Siddiquie B.","key":"e_1_3_2_2_39_1"},{"volume-title":"Efficient Object Category Recognition Using Classemes. In European Conference on Computer Vision. 776--789","author":"Torresani L.","key":"e_1_3_2_2_40_1"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-013-0620-5"},{"volume-title":"Show and Tell: A Neural Image Caption Generator. In IEEE Conference on Computer Vision and Pattern Recognition. 3156--3164","author":"Vinyals O.","key":"e_1_3_2_2_42_1"},{"volume-title":"IEEE Conference on Computer Vision and Pattern Recognition. 6432--6441","author":"Vo N.","key":"e_1_3_2_2_43_1"},{"volume-title":"Localizing and Orienting Street Views Using Overhead Imagery. In European Conference on Computer Vision. 494--509","author":"Vo N. N.","key":"e_1_3_2_2_44_1"},{"volume-title":"Learning Deep Structure-Preserving Image-Text Embeddings. In IEEE Conference on Computer Vision and Pattern Recognition. 5005--5013","author":"Wang L.","key":"e_1_3_2_2_45_1"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/2700292"},{"volume-title":"CAMP: Cross-Modal Adaptive Message Passing for Text-Image Retrieval. In IEEE International Conference on Computer Vision. 5763--5772","author":"Wang Z.","key":"e_1_3_2_2_47_1"},{"volume-title":"Bundling Features for Large Scale Partial-Duplicate Web Image Search. In IEEE Conference on Computer Vision and Pattern Recognition. 25--32","author":"Wu Z.","key":"e_1_3_2_2_48_1"},{"volume-title":"Modeling Context in Referring Expressions. In European Conference on Computer Vision. 69--85","author":"Yu L.","key":"e_1_3_2_2_49_1"},{"volume-title":"Proceedings of the 18th ACM International Conference on Multimedia. 511--520","author":"Zhou W.","key":"e_1_3_2_2_50_1"},{"volume-title":"European Conference on Computer Vision. 391--405","author":"Zitnick C. L.","key":"e_1_3_2_2_51_1"}],"event":{"name":"MM '20: The 28th ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Seattle WA USA","acronym":"MM '20"},"container-title":["Proceedings of the 28th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3394171.3413567","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3394171.3413567","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:47:14Z","timestamp":1750193234000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3394171.3413567"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,10,12]]},"references-count":51,"alternative-id":["10.1145\/3394171.3413567","10.1145\/3394171"],"URL":"https:\/\/doi.org\/10.1145\/3394171.3413567","relation":{},"subject":[],"published":{"date-parts":[[2020,10,12]]},"assertion":[{"value":"2020-10-12","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}