{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,9]],"date-time":"2026-08-09T10:15:24Z","timestamp":1786270524719,"version":"build-2736575974"},"publisher-location":"New York, NY, USA","reference-count":50,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,7,6]],"date-time":"2022-07-06T00:00:00Z","timestamp":1657065600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Natural Science Foundation of China","award":["61976049; 62072080; U20B2063"],"award-info":[{"award-number":["61976049; 62072080; U20B2063"]}]},{"name":"Meituan"},{"name":"Sichuan Science and Technology Program","award":["2019ZDZX0008; 2019YFG0533"],"award-info":[{"award-number":["2019ZDZX0008; 2019YFG0533"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,7,6]]},"DOI":"10.1145\/3477495.3532028","type":"proceedings-article","created":{"date-parts":[[2022,7,7]],"date-time":"2022-07-07T15:12:08Z","timestamp":1657206728000},"page":"960-969","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":16,"title":["Multimodal Disentanglement Variational AutoEncoders for Zero-Shot Cross-Modal Retrieval"],"prefix":"10.1145","author":[{"given":"Jialin","family":"Tian","sequence":"first","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kai","family":"Wang","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xing","family":"Xu","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zuo","family":"Cao","sequence":"additional","affiliation":[{"name":"Meituan, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fumin","family":"Shen","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Heng Tao","family":"Shen","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2022,7,7]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00954"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"crossref","unstructured":"Jingze Chi and Yuxin Peng. 2018a. Dual Adversarial Networks for Zero-shot Cross-media Retrieval. In IJCAI. 663--669. Jingze Chi and Yuxin Peng. 2018a. Dual Adversarial Networks for Zero-shot Cross-media Retrieval. In IJCAI. 663--669.","DOI":"10.24963\/ijcai.2018\/92"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"crossref","unstructured":"Jingze Chi and Yuxin Peng. 2018b. Dual Adversarial Networks for Zero-shot Cross-media Retrieval. In IJCAI. 256--262. Jingze Chi and Yuxin Peng. 2018b. Dual Adversarial Networks for Zero-shot Cross-media Retrieval. In IJCAI. 256--262.","DOI":"10.24963\/ijcai.2018\/92"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2019.2900171"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"crossref","unstructured":"Tat-Seng Chua Jinhui Tang Richang Hong Haojie Li Zhiping Luo and Yan-Tao. Zheng. 2009. NUS-WIDE: A Real-World Web Image Database from National University of Singapore. In ACM CVIR. Tat-Seng Chua Jinhui Tang Richang Hong Haojie Li Zhiping Luo and Yan-Tao. Zheng. 2009. NUS-WIDE: A Real-World Web Image Database from National University of Singapore. In ACM CVIR.","DOI":"10.1145\/1646396.1646452"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00228"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00523"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2019.2923287"},{"key":"e_1_3_2_2_9_1","volume-title":"Style-Guided Zero-Shot Sketch-based Image Retrieval. In British Machine Vision Conference","author":"Dutta Titir","year":"2019","unstructured":"Titir Dutta and Soma Biswas . 2019 b . Style-Guided Zero-Shot Sketch-based Image Retrieval. In British Machine Vision Conference 2019. 209--213. Titir Dutta and Soma Biswas. 2019 b. Style-Guided Zero-Shot Sketch-based Image Retrieval. In British Machine Vision Conference 2019. 209--213."},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cag.2010.07.002"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2010.266"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00087"},{"key":"e_1_3_2_2_13_1","first-page":"143","article-title":"MHTN","volume":"14","author":"Huang Xin","year":"2018","unstructured":"Xin Huang , Yuxin Peng , and Mingkuan Yuan . 2018 . MHTN : Modal-adversarial Hybrid Transfer Network for Cross-modal Retrieval. IEEE Trans. Cybernetics , Vol. 14 , 6 (2018), 143 -- 156 . Xin Huang, Yuxin Peng, and Mingkuan Yuan. 2018. MHTN: Modal-adversarial Hybrid Transfer Network for Cross-modal Retrieval. IEEE Trans. Cybernetics, Vol. 14, 6 (2018), 143--156.","journal-title":"Modal-adversarial Hybrid Transfer Network for Cross-modal Retrieval. IEEE Trans. Cybernetics"},{"key":"e_1_3_2_2_14_1","volume-title":"Advances in Neural Information Processing Systems","volume":"33","author":"Hwang HyeongJoo","year":"2020","unstructured":"HyeongJoo Hwang , Geon-Hyeong Kim , Seunghoon Hong , and Kee-Eung Kim . 2020 . Variational Interaction Information Maximization for Cross-domain Disentanglement . Advances in Neural Information Processing Systems , Vol. 33 (2020). HyeongJoo Hwang, Geon-Hyeong Kim, Seunghoon Hong, and Kee-Eung Kim. 2020. Variational Interaction Information Maximization for Cross-domain Disentanglement. Advances in Neural Information Processing Systems, Vol. 33 (2020)."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2015.2390499"},{"key":"e_1_3_2_2_16_1","volume-title":"Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114","author":"Kingma Diederik P","year":"2013","unstructured":"Diederik P Kingma and Max Welling . 2013. Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114 ( 2013 ). Diederik P Kingma and Max Welling. 2013. Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114 (2013)."},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.473"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01500"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6817"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.247"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00376"},{"key":"e_1_3_2_2_22_1","first-page":"2579","article-title":"Visualizing data using t-SNE","volume":"9","author":"Maaten L.","year":"2008","unstructured":"L. Maaten and G. Hinton . 2008 . Visualizing data using t-SNE . Journal of Machine Learning Research , Vol. 9 (2008), 2579 -- 2605 . L. Maaten and G. Hinton. 2008. Visualizing data using t-SNE. Journal of Machine Learning Research, Vol. 9 (2008), 2579--2605.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2017.2742704"},{"key":"e_1_3_2_2_24_1","volume-title":"NAACL HLT 2010 Workshop on Creating Speech and Language Data with Amazon's Mechanical Turk. 674--686","author":"Rashtchian C.","unstructured":"C. Rashtchian , M. Young , P. Hodosh , and J. Hockenmaier . 2010. Collecting Image Annotations Using Amazon's Mechanical Turk . In NAACL HLT 2010 Workshop on Creating Speech and Language Data with Amazon's Mechanical Turk. 674--686 . C. Rashtchian, M. Young, P. Hodosh, and J. Hockenmaier. 2010. Collecting Image Annotations Using Amazon's Mechanical Turk. In NAACL HLT 2010 Workshop on Creating Speech and Language Data with Amazon's Mechanical Turk. 674--686."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"crossref","unstructured":"N. Rasiwasia J. Costa Pereira E. Coviello G. Doyle G. Lanckriet R. Levy and N. Vasconcelos. 2010. A new approach to cross-modal multimedia retrieval.. In ACM MM. 251--260. N. Rasiwasia J. Costa Pereira E. Coviello G. Doyle G. Lanckriet R. Levy and N. Vasconcelos. 2010. A new approach to cross-modal multimedia retrieval.. In ACM MM. 251--260.","DOI":"10.1145\/1873951.1873987"},{"key":"e_1_3_2_2_26_1","volume-title":"International conference on machine learning. PMLR, 1431--1439","author":"Reed Scott","year":"2014","unstructured":"Scott Reed , Kihyuk Sohn , Yuting Zhang , and Honglak Lee . 2014 . Learning to disentangle factors of variation with manifold interaction . In International conference on machine learning. PMLR, 1431--1439 . Scott Reed, Kihyuk Sohn, Yuting Zhang, and Honglak Lee. 2014. Learning to disentangle factors of variation with manifold interaction. In International conference on machine learning. PMLR, 1431--1439."},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.5244\/C.29.164"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/2897824.2925954"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00379"},{"key":"e_1_3_2_2_30_1","unstructured":"K. Simonyan and A. Zisserman. 2014. Very Deep Convolutional Networks for Large-Scale Image Recognition. CoRR Vol. abs\/1409.1556 (2014). K. Simonyan and A. Zisserman. 2014. Very Deep Convolutional Networks for Large-Scale Image Recognition. CoRR Vol. abs\/1409.1556 (2014)."},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475422"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475676"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"crossref","unstructured":"Bokun Wang Yang Yang Xing Xu Alan Hanjalic and Heng Tao Shen. 2017. Adversarial Cross-Modal Retrieval. In ACM MM. 154--162. Bokun Wang Yang Yang Xing Xu Alan Hanjalic and Heng Tao Shen. 2017. Adversarial Cross-Modal Retrieval. In ACM MM. 154--162.","DOI":"10.1145\/3123266.3123326"},{"key":"e_1_3_2_2_34_1","volume-title":"IEEE International Conference on Computer Vision. 2088--2095","author":"Wang K.","unstructured":"K. Wang , R. He , W. Wang , L. Wang , and T. Tan . 2013. Learning coupled feature spaces for cross-modal matching . In IEEE International Conference on Computer Vision. 2088--2095 . K. Wang, R. He, W. Wang, L. Wang, and T. Tan. 2013. Learning coupled feature spaces for cross-modal matching. In IEEE International Conference on Computer Vision. 2088--2095."},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"crossref","unstructured":"Zhipeng Wang Hao Wang Jiexi Yan Aming Wu and Cheng Deng. 2021. Domain-Smoothing Network for Zero-Shot Sketch-Based Image Retrieval. arxiv: 2106.11841 [cs.CV] Zhipeng Wang Hao Wang Jiexi Yan Aming Wu and Cheng Deng. 2021. Domain-Smoothing Network for Zero-Shot Sketch-Based Image Retrieval. arxiv: 2106.11841 [cs.CV]","DOI":"10.24963\/ijcai.2021\/158"},{"key":"e_1_3_2_2_36_1","first-page":"449","article-title":"Cross-Modal Retrieval With CNN Visual Features: A New Baseline","volume":"47","author":"Wei Yunchao","year":"2017","unstructured":"Yunchao Wei , Yao Zhao , Canyi Lu , Shikui Wei , Luoqi Liu , Zhenfeng Zhu , and Shuicheng Yan . 2017 . Cross-Modal Retrieval With CNN Visual Features: A New Baseline . IEEE Trans. Cybernetics , Vol. 47 , 2 (2017), 449 -- 460 . Yunchao Wei, Yao Zhao, Canyi Lu, Shikui Wei, Luoqi Liu, Zhenfeng Zhu, and Shuicheng Yan. 2017. Cross-Modal Retrieval With CNN Visual Features: A New Baseline. IEEE Trans. Cybernetics, Vol. 47, 2 (2017), 449--460.","journal-title":"IEEE Trans. Cybernetics"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2018.2878970"},{"key":"e_1_3_2_2_38_1","volume-title":"Heng Tao Shen, and Xuelong Li","author":"Xu Xing","year":"2020","unstructured":"Xing Xu , Kaiyi Lin , Lianli Gao , Huimin Lu , Heng Tao Shen, and Xuelong Li . 2020 . Cross-Modal Common Representations by Private-Shared Subspaces Separation. IEEE Transactions on Cybernetics ( 2020), 1--14. Xing Xu, Kaiyi Lin, Lianli Gao, Huimin Lu, Heng Tao Shen, and Xuelong Li. 2020. Cross-Modal Common Representations by Private-Shared Subspaces Separation. IEEE Transactions on Cybernetics (2020), 1--14."},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401149"},{"key":"e_1_3_2_2_40_1","volume-title":"2020 b. Joint Feature Synthesis and Embedding: Adversarial Cross-modal Retrieval Revisited","author":"Xu Xing","year":"2020","unstructured":"Xing Xu , Kaiyi Lin , Yang Yang , Alan Hanjalic , and Heng Tao Shen . 2020 b. Joint Feature Synthesis and Embedding: Adversarial Cross-modal Retrieval Revisited . IEEE Transactions on Pattern Analysis and Machine Intelligence ( 2020 ). Xing Xu, Kaiyi Lin, Yang Yang, Alan Hanjalic, and Heng Tao Shen. 2020 b. Joint Feature Synthesis and Embedding: Adversarial Cross-modal Retrieval Revisited. IEEE Transactions on Pattern Analysis and Machine Intelligence (2020)."},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2019.2928180"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"crossref","unstructured":"Xing Xu Jingkuan Song Huimin Lu Yang Yang Fumin Shen and Zi Huang. 2018. Modal-adversarial Semantic Learning Network for Extendable Cross-modal Retrieval. In ACM ICMR. 46--54. Xing Xu Jingkuan Song Huimin Lu Yang Yang Fumin Shen and Zi Huang. 2018. Modal-adversarial Semantic Learning Network for Extendable Cross-modal Retrieval. In ACM ICMR. 46--54.","DOI":"10.1145\/3206025.3206033"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3401979"},{"key":"e_1_3_2_2_44_1","volume-title":"Proceedings of the Twenty-Ninth International Joint Conference on Artificial Intelligence, IJCAI-20","author":"Xu Xinxun","unstructured":"Xinxun Xu , Muli Yang , Yanhua Yang , and Hao Wang . [n.,d.]. Progressive Domain-Independent Feature Decomposition Network for Zero-Shot Sketch-Based Image Retrieval . In Proceedings of the Twenty-Ninth International Joint Conference on Artificial Intelligence, IJCAI-20 . 984--990. Xinxun Xu, Muli Yang, Yanhua Yang, and Hao Wang. [n.,d.]. Progressive Domain-Independent Feature Decomposition Network for Zero-Shot Sketch-Based Image Retrieval. In Proceedings of the Twenty-Ninth International Joint Conference on Artificial Intelligence, IJCAI-20. 984--990."},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"crossref","unstructured":"F. Yan and K. Mikolajczyk. 2015. Deep correlation for matching images and text. In CVPR. 3441--3450. F. Yan and K. Mikolajczyk. 2015. Deep correlation for matching images and text. In CVPR. 3441--3450.","DOI":"10.1109\/CVPR.2015.7298966"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00947"},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2964319"},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01225-0_19"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.93"},{"key":"e_1_3_2_2_50_1","volume-title":"Sketch-a-net: A deep neural network that beats humans. International journal of computer vision","author":"Yu Qian","year":"2017","unstructured":"Qian Yu , Yongxin Yang , Feng Liu , Yi-Zhe Song , Tao Xiang , and Timothy M Hospedales . 2017 . Sketch-a-net: A deep neural network that beats humans. International journal of computer vision , Vol. 122 , 3 (2017), 411--425. Qian Yu, Yongxin Yang, Feng Liu, Yi-Zhe Song, Tao Xiang, and Timothy M Hospedales. 2017. Sketch-a-net: A deep neural network that beats humans. International journal of computer vision, Vol. 122, 3 (2017), 411--425."}],"event":{"name":"SIGIR '22: The 45th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Madrid Spain","acronym":"SIGIR '22","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 45th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3477495.3532028","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3477495.3532028","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T18:10:21Z","timestamp":1750183821000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3477495.3532028"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,7,6]]},"references-count":50,"alternative-id":["10.1145\/3477495.3532028","10.1145\/3477495"],"URL":"https:\/\/doi.org\/10.1145\/3477495.3532028","relation":{},"subject":[],"published":{"date-parts":[[2022,7,6]]},"assertion":[{"value":"2022-07-07","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}