{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,3]],"date-time":"2026-04-03T21:55:44Z","timestamp":1775253344823,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":55,"publisher":"ACM","license":[{"start":{"date-parts":[[2018,10,15]],"date-time":"2018-10-15T00:00:00Z","timestamp":1539561600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100001381","name":"National Research Foundation Singapore","doi-asserted-by":"publisher","award":["IRC@Singapore Funding Initiative"],"award-info":[{"award-number":["IRC@Singapore Funding Initiative"]}],"id":[{"id":"10.13039\/501100001381","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2018,10,15]]},"DOI":"10.1145\/3240508.3240646","type":"proceedings-article","created":{"date-parts":[[2018,10,18]],"date-time":"2018-10-18T13:52:08Z","timestamp":1539870728000},"page":"1571-1579","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":58,"title":["Interpretable Multimodal Retrieval for Fashion Products"],"prefix":"10.1145","author":[{"given":"Lizi","family":"Liao","sequence":"first","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiangnan","family":"He","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Zhao","sequence":"additional","affiliation":[{"name":"University of British Columbia, Vancouver, Canada"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chong-Wah","family":"Ngo","sequence":"additional","affiliation":[{"name":"City University of Hong Kong, Hong Kong, Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tat-Seng","family":"Chua","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2018,10,15]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.3115\/1225403.1225421"},{"key":"e_1_3_2_1_2_1","volume-title":"Latent dirichlet allocation. JMLR","author":"Blei David M","year":"2003","unstructured":"David M Blei , Andrew Y Ng , and Michael I Jordan . 2003. Latent dirichlet allocation. JMLR ( 2003 ), 993--1022. David M Blei, Andrew Y Ng, and Michael I Jordan. 2003. Latent dirichlet allocation. JMLR (2003), 993--1022."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-33712-3_44"},{"key":"e_1_3_2_1_4_1","volume-title":"AMC: Attention Guided Multi-modal Correlation Learning for Image Search.","author":"Chen Kan","year":"2017","unstructured":"Kan Chen , Trung Bui , Chen Fang , Zhaowen Wang , and Ram Nevatia . 2017 . AMC: Attention Guided Multi-modal Correlation Learning for Image Search. (2017), 6203--6211. Kan Chen, Trung Bui, Chen Fang, Zhaowen Wang, and Ram Nevatia. 2017. AMC: Attention Guided Multi-modal Correlation Learning for Image Search. (2017), 6203--6211."},{"key":"e_1_3_2_1_5_1","volume-title":"IEEE Transactions on robotics and automation","author":"Homem De Mello LS","year":"1990","unstructured":"LS Homem De Mello and Arthur C Sanderson . 1990. AND\/ OR graph representation of assembly plans. IEEE Transactions on robotics and automation ( 1990 ), 188--199. LS Homem De Mello and Arthur C Sanderson. 1990. AND\/OR graph representation of assembly plans. IEEE Transactions on robotics and automation (1990), 188--199."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2011.5995516"},{"key":"e_1_3_2_1_7_1","volume-title":"Imagenet: A large-scale hierarchical image database. In CVPR. 248--255.","author":"Deng Jia","year":"2009","unstructured":"Jia Deng , Wei Dong , Richard Socher , Li-Jia Li , Kai Li , and Li Fei-Fei . 2009 . Imagenet: A large-scale hierarchical image database. In CVPR. 248--255. Jia Deng, Wei Dong, Richard Socher, Li-Jia Li, Kai Li, and Li Fei-Fei. 2009. Imagenet: A large-scale hierarchical image database. In CVPR. 248--255."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2011.5995474"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"crossref","unstructured":"Christiane Fellbaum. 1998. WordNet .Wiley Online Library. Christiane Fellbaum. 1998. WordNet .Wiley Online Library.","DOI":"10.7551\/mitpress\/7287.001.0001"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3178876.3186064"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/2766462.2767780"},{"key":"e_1_3_2_1_12_1","unstructured":"Ian Goodfellow Jean Pouget-Abadie Mehdi Mirza Bing Xu David Warde-Farley Sherjil Ozair Aaron Courville and Yoshua Bengio. 2014. Generative adversarial nets. In NIPS. 2672--2680. Ian Goodfellow Jean Pouget-Abadie Mehdi Mirza Bing Xu David Warde-Farley Sherjil Ozair Aaron Courville and Yoshua Bengio. 2014. Generative adversarial nets. In NIPS. 2672--2680."},{"key":"e_1_3_2_1_13_1","unstructured":"Xintong Han Zuxuan Wu Phoenix X Huang Xiao Zhang Menglong Zhu Yuan Li Yang Zhao and Larry S Davis. 2017. Automatic spatially-aware fashion concept discovery. In ICCV. 1463--1471. Xintong Han Zuxuan Wu Phoenix X Huang Xiao Zhang Menglong Zhu Yuan Li Yang Zhao and Larry S Davis. 2017. Automatic spatially-aware fashion concept discovery. In ICCV. 1463--1471."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"crossref","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep residual learning for image recognition. In CVPR. 770--778. Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep residual learning for image recognition. In CVPR. 770--778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3038912.3052569"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.127"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","unstructured":"Andrej Karpathy and Li Fei-Fei. 2015. Deep visual-semantic alignments for generating image descriptions. In CVPR . 3128--3137. Andrej Karpathy and Li Fei-Fei. 2015. Deep visual-semantic alignments for generating image descriptions. In CVPR . 3128--3137.","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"e_1_3_2_1_18_1","unstructured":"Andrej Karpathy Armand Joulin and Li F Fei-Fei. 2014. Deep fragment embeddings for bidirectional image sentence mapping. In NIPS . 1889--1897. Andrej Karpathy Armand Joulin and Li F Fei-Fei. 2014. Deep fragment embeddings for bidirectional image sentence mapping. In NIPS . 1889--1897."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"crossref","unstructured":"M Hadi Kiapour Kota Yamaguchi Alexander C Berg and Tamara L Berg. 2014. Hipster wars: Discovering elements of fashion styles. In ECCV. 472--488. M Hadi Kiapour Kota Yamaguchi Alexander C Berg and Tamara L Berg. 2014. Hipster wars: Discovering elements of fashion styles. In ECCV. 472--488.","DOI":"10.1007\/978-3-319-10590-1_31"},{"key":"e_1_3_2_1_20_1","volume-title":"Visual Fashion-Product Search at SK Planet. arXiv preprint arXiv:1609.07859","author":"Kim Taewan","year":"2016","unstructured":"Taewan Kim , Seyeong Kim , Sangil Na , Hayoon Kim , Moonki Kim , and Byoung-Ki Jeon . 2016. Visual Fashion-Product Search at SK Planet. arXiv preprint arXiv:1609.07859 ( 2016 ). Taewan Kim, Seyeong Kim, Sangil Na, Hayoon Kim, Moonki Kim, and Byoung-Ki Jeon. 2016. Visual Fashion-Product Search at SK Planet. arXiv preprint arXiv:1609.07859 (2016)."},{"key":"e_1_3_2_1_21_1","volume-title":"Adam: A method for stochastic optimization. In ICLR. 1--15.","author":"Kingma Diederik","year":"2015","unstructured":"Diederik Kingma and Jimmy Ba . 2015 . Adam: A method for stochastic optimization. In ICLR. 1--15. Diederik Kingma and Jimmy Ba. 2015. Adam: A method for stochastic optimization. In ICLR. 1--15."},{"key":"e_1_3_2_1_22_1","unstructured":"Ryan Kiros Ruslan Salakhutdinov and Rich Zemel. 2014. Multimodal neural language models. (2014) 595--603. Ryan Kiros Ruslan Salakhutdinov and Rich Zemel. 2014. Multimodal neural language models. (2014) 595--603."},{"key":"e_1_3_2_1_23_1","volume-title":"Whittlesearch: Image search with relative attribute feedback. In CVPR. 2973--2980.","author":"Kovashka Adriana","year":"2012","unstructured":"Adriana Kovashka , Devi Parikh , and Kristen Grauman . 2012 . Whittlesearch: Image search with relative attribute feedback. In CVPR. 2973--2980. Adriana Kovashka, Devi Parikh, and Kristen Grauman. 2012. Whittlesearch: Image search with relative attribute feedback. In CVPR. 2973--2980."},{"key":"e_1_3_2_1_24_1","unstructured":"Alex Krizhevsky Ilya Sutskever and Geoffrey E Hinton. 2012. Imagenet classification with deep convolutional neural networks. In NIPS . 1097--1105. Alex Krizhevsky Ilya Sutskever and Geoffrey E Hinton. 2012. Imagenet classification with deep convolutional neural networks. In NIPS . 1097--1105."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"crossref","unstructured":"Neeraj Kumar Alexander C Berg Peter N Belhumeur and Shree K Nayar. 2009. Attribute and simile classifiers for face verification. In ICCV. 365--372. Neeraj Kumar Alexander C Berg Peter N Belhumeur and Shree K Nayar. 2009. Attribute and simile classifiers for face verification. In ICCV. 365--372.","DOI":"10.1109\/ICCV.2009.5459250"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3159652.3159716"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","unstructured":"Christoph H Lampert Hannes Nickisch and Stefan Harmeling. 2009. Learning to detect unseen object classes by between-class attribute transfer. In CVPR . 951--958. Christoph H Lampert Hannes Nickisch and Stefan Harmeling. 2009. Learning to detect unseen object classes by between-class attribute transfer. In CVPR . 951--958.","DOI":"10.1109\/CVPR.2009.5206594"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240508.3240605"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.5555\/2354409.2354954"},{"key":"e_1_3_2_1_30_1","volume-title":"Deepfashion: Powering robust clothes recognition and retrieval with rich annotations. In CVPR . 1096--1104.","author":"Liu Ziwei","year":"2016","unstructured":"Ziwei Liu , Ping Luo , Shi Qiu , Xiaogang Wang , and Xiaoou Tang . 2016 . Deepfashion: Powering robust clothes recognition and retrieval with rich annotations. In CVPR . 1096--1104. Ziwei Liu, Ping Luo, Shi Qiu, Xiaogang Wang, and Xiaoou Tang. 2016. Deepfashion: Powering robust clothes recognition and retrieval with rich annotations. In CVPR . 1096--1104."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3159652.3159680"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/2783258.2783381"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/2766462.2767755"},{"key":"e_1_3_2_1_34_1","unstructured":"Tomas Mikolov Ilya Sutskever Kai Chen Greg S Corrado and Jeff Dean. 2013. Distributed representations of words and phrases and their compositionality. In NIPS . 3111--3119. Tomas Mikolov Ilya Sutskever Kai Chen Greg S Corrado and Jeff Dean. 2013. Distributed representations of words and phrases and their compositionality. In NIPS . 3111--3119."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/MMUL.2006.63"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00177"},{"key":"e_1_3_2_1_37_1","volume-title":"high-quality object detection. arXiv preprint arXiv:1412.1441","author":"Szegedy Christian","year":"2014","unstructured":"Christian Szegedy , Scott Reed , Dumitru Erhan , Dragomir Anguelov , and Sergey Ioffe . 2014. Scalable , high-quality object detection. arXiv preprint arXiv:1412.1441 ( 2014 ). Christian Szegedy, Scott Reed, Dumitru Erhan, Dragomir Anguelov, and Sergey Ioffe. 2014. Scalable, high-quality object detection. arXiv preprint arXiv:1412.1441 (2014)."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.websem.2006.06.001"},{"key":"e_1_3_2_1_39_1","volume-title":"Efficient object category recognition using classemes. ECCV","author":"Torresani Lorenzo","year":"2010","unstructured":"Lorenzo Torresani , Martin Szummer , and Andrew Fitzgibbon . 2010. Efficient object category recognition using classemes. ECCV ( 2010 ), 776--789. Lorenzo Torresani, Martin Szummer, and Andrew Fitzgibbon. 2010. Efficient object category recognition using classemes. ECCV (2010), 776--789."},{"key":"e_1_3_2_1_40_1","volume-title":"Disentangling Nonlinear Perceptual Embeddings With Multi-Query Triplet Networks. arXiv preprint arXiv:1603.07810","author":"Veit Andreas","year":"2016","unstructured":"Andreas Veit , Serge Belongie , and Theofanis Karaletsos . 2016. Disentangling Nonlinear Perceptual Embeddings With Multi-Query Triplet Networks. arXiv preprint arXiv:1603.07810 ( 2016 ). Andreas Veit, Serge Belongie, and Theofanis Karaletsos. 2016. Disentangling Nonlinear Perceptual Embeddings With Multi-Query Triplet Networks. arXiv preprint arXiv:1603.07810 (2016)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"crossref","unstructured":"Nakul Verma Dhruv Mahajan Sundararajan Sellamanickam and Vinod Nair. 2012. Learning hierarchical similarity metrics. In CVPR. 2280--2287. Nakul Verma Dhruv Mahajan Sundararajan Sellamanickam and Vinod Nair. 2012. Learning hierarchical similarity metrics. In CVPR. 2280--2287.","DOI":"10.1109\/CVPR.2012.6247938"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"crossref","unstructured":"Liwei Wang Yin Li and Svetlana Lazebnik. 2016. Learning deep structure-preserving image-text embeddings. In CVPR. 5005--5013. Liwei Wang Yin Li and Svetlana Lazebnik. 2016. Learning deep structure-preserving image-text embeddings. In CVPR. 5005--5013.","DOI":"10.1109\/CVPR.2016.541"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3159652.3159710"},{"key":"e_1_3_2_1_44_1","unstructured":"Kilian Q Weinberger John Blitzer and Lawrence K Saul. 2006. Distance metric learning for large margin nearest neighbor classification. In NIPS . 1473--1480. Kilian Q Weinberger John Blitzer and Lawrence K Saul. 2006. Distance metric learning for large margin nearest neighbor classification. In NIPS . 1473--1480."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3038912.3052558"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"crossref","unstructured":"Kota Yamaguchi M Hadi Kiapour Luis E Ortiz and Tamara L Berg. 2012. Parsing clothing in fashion photographs. In CVPR. 3570--3577. Kota Yamaguchi M Hadi Kiapour Luis E Ortiz and Tamara L Berg. 2012. Parsing clothing in fashion photographs. In CVPR. 3570--3577.","DOI":"10.1109\/CVPR.2012.6248101"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3097983.3098162"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"crossref","unstructured":"Hanwang Zhang Zawlin Kyaw Shih-Fu Chang and Tat-Seng Chua. 2017. Visual translation embedding network for visual relation detection. In CVPR . 5532--5540. Hanwang Zhang Zawlin Kyaw Shih-Fu Chang and Tat-Seng Chua. 2017. Visual translation embedding network for visual relation detection. In CVPR . 5532--5540.","DOI":"10.1109\/CVPR.2017.331"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/2502081.2502093"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/1835804.1835930"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.212"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"crossref","unstructured":"Bo Zhao Jiashi Feng Xiao Wu and Shuicheng Yan. 2017. Memory-augmented attribute manipulation networks for interactive fashion search. In CVPR . 1520--1528. Bo Zhao Jiashi Feng Xiao Wu and Shuicheng Yan. 2017. Memory-augmented attribute manipulation networks for interactive fashion search. In CVPR . 1520--1528.","DOI":"10.1109\/CVPR.2017.652"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"crossref","unstructured":"Bolei Zhou Aditya Khosla Agata Lapedriza Aude Oliva and Antonio Torralba. 2016a. Learning deep features for discriminative localization. In CVPR . 2921--2929. Bolei Zhou Aditya Khosla Agata Lapedriza Aude Oliva and Antonio Torralba. 2016a. Learning deep features for discriminative localization. In CVPR . 2921--2929.","DOI":"10.1109\/CVPR.2016.319"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P16-2034"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"crossref","unstructured":"Jun-Yan Zhu Philipp Kr\"ahenb\u00fchl Eli Shechtman and Alexei A Efros. 2016. Generative visual manipulation on the natural image manifold. In ECCV. 597--613. Jun-Yan Zhu Philipp Kr\"ahenb\u00fchl Eli Shechtman and Alexei A Efros. 2016. Generative visual manipulation on the natural image manifold. In ECCV. 597--613.","DOI":"10.1007\/978-3-319-46454-1_36"}],"event":{"name":"MM '18: ACM Multimedia Conference","location":"Seoul Republic of Korea","acronym":"MM '18","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 26th ACM international conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3240508.3240646","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3240508.3240646","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,3]],"date-time":"2026-04-03T20:40:48Z","timestamp":1775248848000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3240508.3240646"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,10,15]]},"references-count":55,"alternative-id":["10.1145\/3240508.3240646","10.1145\/3240508"],"URL":"https:\/\/doi.org\/10.1145\/3240508.3240646","relation":{},"subject":[],"published":{"date-parts":[[2018,10,15]]},"assertion":[{"value":"2018-10-15","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}