{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,26]],"date-time":"2026-02-26T15:42:27Z","timestamp":1772120547995,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":73,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,10,17]],"date-time":"2021-10-17T00:00:00Z","timestamp":1634428800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61972349"],"award-info":[{"award-number":["61972349"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2018YFB1403202"],"award-info":[{"award-number":["2018YFB1403202"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,10,17]]},"DOI":"10.1145\/3474085.3475619","type":"proceedings-article","created":{"date-parts":[[2021,10,18]],"date-time":"2021-10-18T06:09:05Z","timestamp":1634537345000},"page":"4600-4609","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":27,"title":["Image Search with Text Feedback by Deep Hierarchical Attention Mutual Information Maximization"],"prefix":"10.1145","author":[{"given":"Chunbin","family":"Gu","sequence":"first","affiliation":[{"name":"Zhejiang University &amp; Alibaba-Zhejiang University Joint Institute of Frontier Technologies, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiajun","family":"Bu","sequence":"additional","affiliation":[{"name":"Zhejiang University &amp; Alibaba-Zhejiang University Joint Institute of Frontier Technologies, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhen","family":"Zhang","sequence":"additional","affiliation":[{"name":"Zhejiang UniversityZhejiang University &amp; Alibaba-Zhejiang University Joint Institute of Frontier Technologies, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhi","family":"Yu","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dongfang","family":"Ma","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Wang","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,10,17]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00804"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.5555\/3454287.3455679"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.5555\/193138"},{"key":"e_1_3_2_2_4_1","volume-title":"Mutual information maximization: models of cortical self-organization. Network: Computation in neural systems 7, 1","author":"Becker Suzanna","year":"1996","unstructured":"Suzanna Becker . 1996. Mutual information maximization: models of cortical self-organization. Network: Computation in neural systems 7, 1 ( 1996 ), 7--31. Suzanna Becker. 1996. Mutual information maximization: models of cortical self-organization. Network: Computation in neural systems 7, 1 (1996), 7--31."},{"key":"e_1_3_2_2_5_1","volume-title":"Mine: mutual information neural estimation. arXiv preprint arXiv:1801.04062","author":"Belghazi Mohamed Ishmael","year":"2018","unstructured":"Mohamed Ishmael Belghazi , Aristide Baratin , Sai Rajeswar , Sherjil Ozair , Yoshua Bengio , Aaron Courville , and R Devon Hjelm . 2018. Mine: mutual information neural estimation. arXiv preprint arXiv:1801.04062 ( 2018 ). Mohamed Ishmael Belghazi, Aristide Baratin, Sai Rajeswar, Sherjil Ozair, Yoshua Bengio, Aaron Courville, and R Devon Hjelm. 2018. Mine: mutual information neural estimation. arXiv preprint arXiv:1801.04062 (2018)."},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.5555\/1886063.1886114"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00196"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2018.10.082"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3465055"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"crossref","unstructured":"Yanbei Chen and Loris Bazzani. 2020. Learning Joint Visual Semantic Matching Embeddings for Language-guided Retrieval. ECCV.  Yanbei Chen and Loris Bazzani. 2020. Learning Joint Visual Semantic Matching Embeddings for Language-guided Retrieval. ECCV.","DOI":"10.1007\/978-3-030-58542-6_9"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00307"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2005.202"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_2_14_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin , Ming-Wei Chang , Kenton Lee , and Kristina Toutanova . 2018 . Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018). Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1002\/cpa.3160360204"},{"key":"e_1_3_2_2_16_1","volume-title":"Jamie Ryan Kiros, and Sanja Fidler","author":"Faghri Fartash","year":"2017","unstructured":"Fartash Faghri , David J Fleet , Jamie Ryan Kiros, and Sanja Fidler . 2017 . Vse++: Improving visual-semantic embeddings with hard negatives. arXiv preprint arXiv:1707.05612 (2017). Fartash Faghri, David J Fleet, Jamie Ryan Kiros, and Sanja Fidler. 2017. Vse++: Improving visual-semantic embeddings with hard negatives. arXiv preprint arXiv:1707.05612 (2017)."},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00680"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.5555\/2969033.2969125"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46466-4_15"},{"key":"e_1_3_2_2_20_1","volume-title":"Cross-modal Image Retrieval with Deep Mutual Information Maximization. arXiv preprint arXiv:2103.06032","author":"Gu Chunbin","year":"2021","unstructured":"Chunbin Gu , Jiajun Bu , Xixi Zhou , Chengwei Yao , Dongfang Ma , Zhi Yu , and Xifeng Yan . 2021. Cross-modal Image Retrieval with Deep Mutual Information Maximization. arXiv preprint arXiv:2103.06032 ( 2021 ). Chunbin Gu, Jiajun Bu, Xixi Zhou, Chengwei Yao, Dongfang Ma, Zhi Yu, and Xifeng Yan. 2021. Cross-modal Image Retrieval with Deep Mutual Information Maximization. arXiv preprint arXiv:2103.06032 (2021)."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3351053"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.5555\/3326943.3327006"},{"key":"e_1_3_2_2_23_1","volume-title":"Fashion IQ: A New Dataset towards Retrieving Images by Natural Language Feedback. arXiv preprint arXiv:1905.12794","author":"Guo Xiaoxiao","year":"2019","unstructured":"Xiaoxiao Guo , Hui Wu , Yupeng Gao , Steven Rennie , and Rogerio Feris . 2019 . Fashion IQ: A New Dataset towards Retrieving Images by Natural Language Feedback. arXiv preprint arXiv:1905.12794 (2019). Xiaoxiao Guo, Hui Wu, Yupeng Gao, Steven Rennie, and Rogerio Feris. 2019. Fashion IQ: A New Dataset towards Retrieving Images by Natural Language Feedback. arXiv preprint arXiv:1905.12794 (2019)."},{"key":"e_1_3_2_2_24_1","volume-title":"Proceedings of the Thirteenth International Conference on Artificial Intelligence and Statistics. 297--304","author":"Gutmann Michael","year":"2010","unstructured":"Michael Gutmann and Aapo Hyv\u00e4rinen . 2010 . Noise-contrastive estimation: A new estimation principle for unnormalized statistical models . In Proceedings of the Thirteenth International Conference on Artificial Intelligence and Statistics. 297--304 . Michael Gutmann and Aapo Hyv\u00e4rinen. 2010. Noise-contrastive estimation: A new estimation principle for unnormalized statistical models. In Proceedings of the Thirteenth International Conference on Artificial Intelligence and Statistics. 297--304."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.5555\/2503308.2188396"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.163"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_2_28_1","volume-title":"Learning deep representations by mutual information estimation and maximization. arXiv preprint arXiv:1808.06670","author":"Hjelm R Devon","year":"2018","unstructured":"R Devon Hjelm , Alex Fedorov , Samuel Lavoie-Marchildon , Karan Grewal , Phil Bachman , Adam Trischler , and Yoshua Bengio . 2018. Learning deep representations by mutual information estimation and maximization. arXiv preprint arXiv:1808.06670 ( 2018 ). R Devon Hjelm, Alex Fedorov, Samuel Lavoie-Marchildon, Karan Grewal, Phil Bachman, Adam Trischler, and Yoshua Bengio. 2018. Learning deep representations by mutual information estimation and maximization. arXiv preprint arXiv:1808.06670 (2018)."},{"key":"e_1_3_2_2_29_1","volume-title":"Mobilenets: Efficient convolutional neural networks for mobile vision applications. arXiv preprint arXiv:1704.04861","author":"Howard Andrew G","year":"2017","unstructured":"Andrew G Howard , Menglong Zhu , Bo Chen , Dmitry Kalenichenko , Weijun Wang , Tobias Weyand , Marco Andreetto , and Hartwig Adam . 2017 . Mobilenets: Efficient convolutional neural networks for mobile vision applications. arXiv preprint arXiv:1704.04861 (2017). Andrew G Howard, Menglong Zhu, Bo Chen, Dmitry Kalenichenko, Weijun Wang, Tobias Weyand, Marco Andreetto, and Hartwig Adam. 2017. Mobilenets: Efficient convolutional neural networks for mobile vision applications. arXiv preprint arXiv:1704.04861 (2017)."},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"crossref","unstructured":"Jie Hu Li Shen and Gang Sun. 2018. Squeeze-and-excitation networks. In Proceed- ings of the IEEE conference on computer vision and pattern recognition. 7132--7141.  Jie Hu Li Shen and Gang Sun. 2018. Squeeze-and-excitation networks. In Proceed- ings of the IEEE conference on computer vision and pattern recognition. 7132--7141.","DOI":"10.1109\/CVPR.2018.00745"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.5555\/3367032.3367145"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00473"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2020.2992393"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.5555\/3157096.3157137"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.1309933111"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2013.44"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.5555\/2354409.2354723"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.124"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.5555\/3157096.3157129"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01246-5_11"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00637"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.11"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.5555\/3157096.3157127"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00077"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2011.6126281"},{"key":"e_1_3_2_2_46_1","volume-title":"International Conference on Machine Learning. PMLR, 4055--4064","author":"Parmar Niki","year":"2018","unstructured":"Niki Parmar , Ashish Vaswani , Jakob Uszkoreit , Lukasz Kaiser , Noam Shazeer , Alexander Ku , and Dustin Tran . 2018 . Image transformer . In International Conference on Machine Learning. PMLR, 4055--4064 . Niki Parmar, Ashish Vaswani, Jakob Uszkoreit, Lukasz Kaiser, Noam Shazeer, Alexander Ku, and Dustin Tran. 2018. Image transformer. In International Conference on Machine Learning. PMLR, 4055--4064."},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11671"},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.5555\/3454287.3454294"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/76.718510"},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/2897824.2925954"},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.5555\/3295222.3295250"},{"key":"e_1_3_2_2_52_1","volume-title":"German Conference on Pattern Recognition. Springer, 228--243","author":"Sayed Nawid","year":"2018","unstructured":"Nawid Sayed , Biagio Brattoli , and Bj\u00f6rn Ommer . 2018 . Cross and learn: Cross- modal self-supervision . In German Conference on Pattern Recognition. Springer, 228--243 . Nawid Sayed, Biagio Brattoli, and Bj\u00f6rn Ommer. 2018. Cross and learn: Cross- modal self-supervision. In German Conference on Pattern Recognition. Springer, 228--243."},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298682"},{"key":"e_1_3_2_2_54_1","volume-title":"Self-attention with relative position representations. arXiv preprint arXiv:1803.02155","author":"Shaw Peter","year":"2018","unstructured":"Peter Shaw , Jakob Uszkoreit , and Ashish Vaswani . 2018. Self-attention with relative position representations. arXiv preprint arXiv:1803.02155 ( 2018 ). Peter Shaw, Jakob Uszkoreit, and Ashish Vaswani. 2018. Self-attention with relative position representations. arXiv preprint arXiv:1803.02155 (2018)."},{"key":"e_1_3_2_2_55_1","volume-title":"Opening the black box of deep neural networks via information. arXiv preprint arXiv:1703.00810","author":"Shwartz-Ziv Ravid","year":"2017","unstructured":"Ravid Shwartz-Ziv and Naftali Tishby . 2017. Opening the black box of deep neural networks via information. arXiv preprint arXiv:1703.00810 ( 2017 ). Ravid Shwartz-Ziv and Naftali Tishby. 2017. Opening the black box of deep neural networks via information. arXiv preprint arXiv:1703.00810 (2017)."},{"key":"e_1_3_2_2_56_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00177"},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.244"},{"key":"e_1_3_2_2_58_1","doi-asserted-by":"publisher","DOI":"10.1007\/s13735-012-0014-4"},{"key":"e_1_3_2_2_59_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2019.10.073"},{"key":"e_1_3_2_2_60_1","volume-title":"Contrastive multiview coding. arXiv preprint arXiv:1906.05849","author":"Tian Yonglong","year":"2019","unstructured":"Yonglong Tian , Dilip Krishnan , and Phillip Isola . 2019. Contrastive multiview coding. arXiv preprint arXiv:1906.05849 ( 2019 ). Yonglong Tian, Dilip Krishnan, and Phillip Isola. 2019. Contrastive multiview coding. arXiv preprint arXiv:1906.05849 (2019)."},{"key":"e_1_3_2_2_61_1","doi-asserted-by":"publisher","DOI":"10.5555\/3295222.3295349"},{"key":"e_1_3_2_2_62_1","volume-title":"Hien Van Nguyen, and S Kevin Zhou","author":"Vemulapalli Raviteja","year":"2017","unstructured":"Raviteja Vemulapalli , Hien Van Nguyen, and S Kevin Zhou . 2017 . Deep networks and mutual information maximization for cross-modal medical image synthesis. In Deep Learning for Medical Image Analysis. Elsevier , 381--403. Raviteja Vemulapalli, Hien Van Nguyen, and S Kevin Zhou. 2017. Deep networks and mutual information maximization for cross-modal medical image synthesis. In Deep Learning for Medical Image Analysis. Elsevier, 381--403."},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00660"},{"key":"e_1_3_2_2_64_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123326"},{"key":"e_1_3_2_2_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01184"},{"key":"e_1_3_2_2_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.541"},{"key":"e_1_3_2_2_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00813"},{"key":"e_1_3_2_2_68_1","doi-asserted-by":"publisher","DOI":"10.1162\/089976602317318938"},{"key":"e_1_3_2_2_69_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00080"},{"key":"e_1_3_2_2_70_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00644"},{"key":"e_1_3_2_2_71_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413917"},{"key":"e_1_3_2_2_72_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.652"},{"key":"e_1_3_2_2_73_1","volume-title":"Relevance feedback in image retrieval: A comprehensive review. Multimedia systems 8, 6","author":"Zhou Xiang Sean","year":"2003","unstructured":"Xiang Sean Zhou and Thomas S Huang . 2003. Relevance feedback in image retrieval: A comprehensive review. Multimedia systems 8, 6 ( 2003 ), 536--544 Xiang Sean Zhou and Thomas S Huang. 2003. Relevance feedback in image retrieval: A comprehensive review. Multimedia systems 8, 6 (2003), 536--544"}],"event":{"name":"MM '21: ACM Multimedia Conference","location":"Virtual Event China","acronym":"MM '21","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 29th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3475619","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3474085.3475619","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:48:24Z","timestamp":1750193304000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3475619"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,10,17]]},"references-count":73,"alternative-id":["10.1145\/3474085.3475619","10.1145\/3474085"],"URL":"https:\/\/doi.org\/10.1145\/3474085.3475619","relation":{},"subject":[],"published":{"date-parts":[[2021,10,17]]},"assertion":[{"value":"2021-10-17","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}