{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:34:21Z","timestamp":1750221261947,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":33,"publisher":"ACM","license":[{"start":{"date-parts":[[2017,10,23]],"date-time":"2017-10-23T00:00:00Z","timestamp":1508716800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National High Technology Research and Development Program of China","award":["2015AA016004"],"award-info":[{"award-number":["2015AA016004"]}]},{"name":"National Natural Science Foundation of China","award":["61370126","U1636211","61672081","61602237"],"award-info":[{"award-number":["61370126","U1636211","61672081","61602237"]}]},{"name":"Fund of the State Key Laboratory of Software Development Environment","award":["SKLSDE-2017ZX-19"],"award-info":[{"award-number":["SKLSDE-2017ZX-19"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2017,10,23]]},"DOI":"10.1145\/3126686.3126720","type":"proceedings-article","created":{"date-parts":[[2017,10,23]],"date-time":"2017-10-23T19:20:32Z","timestamp":1508786432000},"page":"460-468","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":14,"title":["Learning Social Image Embedding with Deep Multimodal Attention Networks"],"prefix":"10.1145","author":[{"given":"Feiran","family":"Huang","sequence":"first","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaoming","family":"Zhang","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhoujun","family":"Li","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tao","family":"Mei","sequence":"additional","affiliation":[{"name":"Microsoft Research, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yueying","family":"He","sequence":"additional","affiliation":[{"name":"National Computer Network Emergency Response Technical Team\/Coordination Center of China, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhonghua","family":"Zhao","sequence":"additional","affiliation":[{"name":"National Computer Network Emergency Response Technical Team\/Coordination Center of China, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2017,10,23]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Proceedings of the 30th International Conference on Machine Learning, ICML 2013, Atlanta, GA, USA, 16-21 June 2013 (JMLR Workshop and Conference Proceedings)","volume":"28","author":"Andrew Galen","year":"2013","unstructured":"Galen Andrew , Raman Arora , Jeff A. Bilmes , and Karen Livescu . 2013 . Deep Canonical Correlation Analysis . In Proceedings of the 30th International Conference on Machine Learning, ICML 2013, Atlanta, GA, USA, 16-21 June 2013 (JMLR Workshop and Conference Proceedings) , Vol. 28 . JMLR.org, 1247--1255. http:\/\/jmlr.org\/proceedings\/papers\/v28\/andrew13.html Galen Andrew, Raman Arora, Jeff A. Bilmes, and Karen Livescu. 2013. Deep Canonical Correlation Analysis. In Proceedings of the 30th International Conference on Machine Learning, ICML 2013, Atlanta, GA, USA, 16-21 June 2013 (JMLR Workshop and Conference Proceedings), Vol. 28. JMLR.org, 1247--1255. http:\/\/jmlr.org\/proceedings\/papers\/v28\/andrew13.html"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/2806416.2806512"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/2783258.2783296"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/1646396.1646452"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-009-0275-4"},{"key":"e_1_3_2_1_7_1","volume-title":"Advances in Neural Information Processing Systems 26: 27th Annual Conference on Neural Information Processing Systems","author":"Frome Andrea","year":"2013","unstructured":"Andrea Frome , Gregory S. Corrado , Jonathon Shlens , Samy Bengio , Jeffrey Dean , Marc'Aurelio Ranzato , and Tomas Mikolov . 2013. DeViSE: A Deep Visual-Semantic Embedding Model . In Advances in Neural Information Processing Systems 26: 27th Annual Conference on Neural Information Processing Systems 2013 . Proceedings of a meeting held December 5-8, 2013, Lake Tahoe, Nevada, United States., Christopher J. C. Burges, Leon Bottou, Zoubin Ghahramani, and Kilian Q. Weinberger (Eds .). 2121--2129. http:\/\/papers.nips.cc\/paper\/5204-devise-a-deep-visual-semantic-embedding-model Andrea Frome, Gregory S. Corrado, Jonathon Shlens, Samy Bengio, Jeffrey Dean, Marc'Aurelio Ranzato, and Tomas Mikolov. 2013. DeViSE: A Deep Visual-Semantic Embedding Model. In Advances in Neural Information Processing Systems 26: 27th Annual Conference on Neural Information Processing Systems 2013. Proceedings of a meeting held December 5-8, 2013, Lake Tahoe, Nevada, United States., Christopher J. C. Burges, Leon Bottou, Zoubin Ghahramani, and Kilian Q. Weinberger (Eds.). 2121--2129. http:\/\/papers.nips.cc\/paper\/5204-devise-a-deep-visual-semantic-embedding-model"},{"key":"e_1_3_2_1_8_1","volume-title":"2016 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2016","author":"Gustavo Carneiro Vijay Kumar B. G","year":"2016","unstructured":"Vijay Kumar B. G , Gustavo Carneiro , and Ian D. Reid . 2016. Learning Local Image Descriptors with Deep Siamese and Triplet Convolutional Networks by Minimizing Global Loss Functions . In 2016 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2016 , Las Vegas, NV, USA , June 27-30, 2016 . IEEE Computer Society, 5385--5394. Vijay Kumar B. G, Gustavo Carneiro, and Ian D. Reid. 2016. Learning Local Image Descriptors with Deep Siamese and Triplet Convolutional Networks by Minimizing Global Loss Functions. In 2016 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2016, Las Vegas, NV, USA, June 27-30, 2016. IEEE Computer Society, 5385--5394."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-013-0658-4"},{"key":"e_1_3_2_1_10_1","volume-title":"Computer Vision - ECCV 2014 - 13th European Conference","author":"Gong Yunchao","year":"2014","unstructured":"Yunchao Gong , Liwei Wang , Micah Hodosh , Julia Hockenmaier , and Svetlana Lazebnik . 2014. Improving Image-Sentence Embeddings Using Large Weakly Annotated Photo Collections . In Computer Vision - ECCV 2014 - 13th European Conference , Zurich, Switzerland, September 6-12, 2014 , Proceedings, Part IV (Lecture Notes in Computer Science), David J. Fleet, Tomas Pajdla, Bernt Schiele, and Tinne Tuytelaars (Eds.), Vol. 8692 . Springer , 529--545. Yunchao Gong, Liwei Wang, Micah Hodosh, Julia Hockenmaier, and Svetlana Lazebnik. 2014. Improving Image-Sentence Embeddings Using Large Weakly Annotated Photo Collections. In Computer Vision - ECCV 2014 - 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part IV (Lecture Notes in Computer Science), David J. Fleet, Tomas Pajdla, Bernt Schiele, and Tinne Tuytelaars (Eds.), Vol. 8692. Springer, 529--545."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1162\/0899766042321814"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/1460096.1460104"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"e_1_3_2_1_14_1","volume-title":"Fisher Vectors Derived from Hybrid Gaussian-Laplacian Mixture Models for Image Annotation. CoRR abs\/1411.7399","author":"Klein Benjamin","year":"2014","unstructured":"Benjamin Klein , Guy Lev , Gil Sadeh , and Lior Wolf . 2014. Fisher Vectors Derived from Hybrid Gaussian-Laplacian Mixture Models for Image Annotation. CoRR abs\/1411.7399 ( 2014 ). http:\/\/arxiv.org\/abs\/1411.7399 Benjamin Klein, Guy Lev, Gil Sadeh, and Lior Wolf. 2014. Fisher Vectors Derived from Hybrid Gaussian-Laplacian Mixture Models for Image Annotation. CoRR abs\/1411.7399 (2014). http:\/\/arxiv.org\/abs\/1411.7399"},{"key":"e_1_3_2_1_15_1","volume-title":"DASFAA 2017, Suzhou, China, March 27-30, 2017, Proceedings, Part I (Lecture Notes in Computer Science), K. Sel\u00e7uk Candan, Lei Chen, Torben Bach Pedersen, Lijun Chang, and Wen Hua (Eds.)","volume":"10177","author":"Li Chaozhuo","year":"2017","unstructured":"Chaozhuo Li , Senzhang Wang , Dejian Yang , Zhoujun Li , Yang Yang , Xiaoming Zhang , and Jianshe Zhou . 2017 . PPNE: Property Preserving Network Embedding. In Database Systems for Advanced Applications - 22nd International Conference , DASFAA 2017, Suzhou, China, March 27-30, 2017, Proceedings, Part I (Lecture Notes in Computer Science), K. Sel\u00e7uk Candan, Lei Chen, Torben Bach Pedersen, Lijun Chang, and Wen Hua (Eds.) , Vol. 10177 . Springer, 163--179. Chaozhuo Li, Senzhang Wang, Dejian Yang, Zhoujun Li, Yang Yang, Xiaoming Zhang, and Jianshe Zhou. 2017. PPNE: Property Preserving Network Embedding. In Database Systems for Advanced Applications - 22nd International Conference, DASFAA 2017, Suzhou, China, March 27-30, 2017, Proceedings, Part I (Lecture Notes in Computer Science), K. Sel\u00e7uk Candan, Lei Chen, Torben Bach Pedersen, Lijun Chang, and Wen Hua (Eds.), Vol. 10177. Springer, 163--179."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/2733373.2806247"},{"key":"e_1_3_2_1_17_1","volume-title":"Yuille","author":"Mao Junhua","year":"2014","unstructured":"Junhua Mao , Wei Xu , Yi Yang , Jiang Wang , and Alan L . Yuille . 2014 . Deep Captioning with Multimodal Recurrent Neural Networks (m-RNN). CoRR abs\/1412.6632 (2014). http:\/\/arxiv.org\/abs\/1412.6632 Junhua Mao, Wei Xu, Yi Yang, Jiang Wang, and Alan L. Yuille. 2014. Deep Captioning with Multimodal Recurrent Neural Networks (m-RNN). CoRR abs\/1412.6632 (2014). http:\/\/arxiv.org\/abs\/1412.6632"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-33765-9_59"},{"key":"e_1_3_2_1_19_1","volume-title":"Zero-Shot Learning by Convex Combination of Semantic Embeddings. CoRR abs\/1312.5650","author":"Norouzi Mohammad","year":"2013","unstructured":"Mohammad Norouzi , Tomas Mikolov , Samy Bengio , Yoram Singer , Jonathon Shlens , Andrea Frome , Greg Corrado , and Jeffrey Dean . 2013. Zero-Shot Learning by Convex Combination of Semantic Embeddings. CoRR abs\/1312.5650 ( 2013 ). http:\/\/arxiv.org\/abs\/1312.5650 Mohammad Norouzi, Tomas Mikolov, Samy Bengio, Yoram Singer, Jonathon Shlens, Andrea Frome, Greg Corrado, and Jeffrey Dean. 2013. Zero-Shot Learning by Convex Combination of Semantic Embeddings. CoRR abs\/1312.5650 (2013). http:\/\/arxiv.org\/abs\/1312.5650"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/2623330.2623732"},{"key":"e_1_3_2_1_21_1","volume-title":"Proceedings of the Twelfth SIAM International Conference on Data Mining","author":"Qi Guo-Jun","year":"2012","unstructured":"Guo-Jun Qi , Charu C. Aggarwal , and Thomas S. Huang . 2012. Transfer Learning of Distance Metrics by Cross-Domain Metric Sampling across Heterogeneous Spaces . In Proceedings of the Twelfth SIAM International Conference on Data Mining , Anaheim, California, USA , April 26-28, 2012 . SIAM \/ Omnipress, 528--539. Guo-Jun Qi, Charu C. Aggarwal, and Thomas S. Huang. 2012. Transfer Learning of Distance Metrics by Cross-Domain Metric Sampling across Heterogeneous Spaces. In Proceedings of the Twelfth SIAM International Conference on Data Mining, Anaheim, California, USA, April 26-28, 2012. SIAM \/ Omnipress, 528--539."},{"key":"e_1_3_2_1_22_1","volume-title":"Exploring Models and Data for Image Question Answering. In Advances in Neural Information Processing Systems 28: Annual Conference on Neural Information Processing Systems 2015","author":"Ren Mengye","year":"2015","unstructured":"Mengye Ren , Ryan Kiros , and Richard S. Zemel . 2015 . Exploring Models and Data for Image Question Answering. In Advances in Neural Information Processing Systems 28: Annual Conference on Neural Information Processing Systems 2015 , December 7-12, 2015 , Montreal, Quebec, Canada, Corinna Cortes, Neil D. Lawrence, Daniel D. Lee, Masashi Sugiyama, and Roman Garnett (Eds.). 2953--2961. http:\/\/papers.nips.cc\/paper\/5640-exploring-models-and-data-for-image-question-answering Mengye Ren, Ryan Kiros, and Richard S. Zemel. 2015. Exploring Models and Data for Image Question Answering. In Advances in Neural Information Processing Systems 28: Annual Conference on Neural Information Processing Systems 2015, December 7-12, 2015, Montreal, Quebec, Canada, Corinna Cortes, Neil D. Lawrence, Daniel D. Lee, Masashi Sugiyama, and Roman Garnett (Eds.). 2953--2961. http:\/\/papers.nips.cc\/paper\/5640-exploring-models-and-data-for-image-question-answering"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2967212"},{"key":"e_1_3_2_1_24_1","volume-title":"Very Deep Convolutional Networks for Large-Scale Image Recognition. CoRR abs\/1409.1556","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman . 2014. Very Deep Convolutional Networks for Large-Scale Image Recognition. CoRR abs\/1409.1556 ( 2014 ). http:\/\/arxiv.org\/abs\/1409.1556 Karen Simonyan and Andrew Zisserman. 2014. Very Deep Convolutional Networks for Large-Scale Image Recognition. CoRR abs\/1409.1556 (2014). http:\/\/arxiv.org\/abs\/1409.1556"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/2736277.2741093"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/2939672.2939753"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.320"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298930"},{"key":"e_1_3_2_1_29_1","volume-title":"A Unified View of Multi-Label Performance Measures. CoRR abs\/1609.00288","author":"Wu Xi-Zhu","year":"2016","unstructured":"Xi-Zhu Wu and Zhi-Hua Zhou . 2016. A Unified View of Multi-Label Performance Measures. CoRR abs\/1609.00288 ( 2016 ). http:\/\/arxiv.org\/abs\/1609.00288 Xi-Zhu Wu and Zhi-Hua Zhou. 2016. A Unified View of Multi-Label Performance Measures. CoRR abs\/1609.00288 (2016). http:\/\/arxiv.org\/abs\/1609.00288"},{"key":"e_1_3_2_1_30_1","volume-title":"Proceedings of the 32nd International Conference on Machine Learning, ICML 2015, Lille, France, 6-11 July 2015 (JMLR Workshop and Conference Proceedings), Francis R. Bach and David M. Blei (Eds.)","volume":"37","author":"Xu Kelvin","year":"2015","unstructured":"Kelvin Xu , Jimmy Ba , Ryan Kiros , Kyunghyun Cho , Aaron C. Courville , Ruslan Salakhutdinov , Richard S. Zemel , and Yoshua Bengio . 2015 . Show, Attend and Tell: Neural Image Caption Generation with Visual Attention . In Proceedings of the 32nd International Conference on Machine Learning, ICML 2015, Lille, France, 6-11 July 2015 (JMLR Workshop and Conference Proceedings), Francis R. Bach and David M. Blei (Eds.) , Vol. 37 . JMLR.org, 2048--2057. http:\/\/jmlr.org\/proceedings\/papers\/v37\/xuc15.html Kelvin Xu, Jimmy Ba, Ryan Kiros, Kyunghyun Cho, Aaron C. Courville, Ruslan Salakhutdinov, Richard S. Zemel, and Yoshua Bengio. 2015. Show, Attend and Tell: Neural Image Caption Generation with Visual Attention. In Proceedings of the 32nd International Conference on Machine Learning, ICML 2015, Lille, France, 6-11 July 2015 (JMLR Workshop and Conference Proceedings), Francis R. Bach and David M. Blei (Eds.), Vol. 37. JMLR.org, 2048--2057. http:\/\/jmlr.org\/proceedings\/papers\/v37\/xuc15.html"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298966"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.31193\/ssap.01.9787509791011"},{"key":"e_1_3_2_1_33_1","volume-title":"Learning Spatial Regularization with Image-level Supervisions for Multi-label Image Classification. CoRR abs\/1702.05891","author":"Zhu Feng","year":"2017","unstructured":"Feng Zhu , Hongsheng Li , Wanli Ouyang , Nenghai Yu , and Xiaogang Wang . 2017. Learning Spatial Regularization with Image-level Supervisions for Multi-label Image Classification. CoRR abs\/1702.05891 ( 2017 ). http:\/\/arxiv.org\/abs\/1702.05891 Feng Zhu, Hongsheng Li, Wanli Ouyang, Nenghai Yu, and Xiaogang Wang. 2017. Learning Spatial Regularization with Image-level Supervisions for Multi-label Image Classification. CoRR abs\/1702.05891 (2017). http:\/\/arxiv.org\/abs\/1702.05891"}],"event":{"name":"MM '17: ACM Multimedia Conference","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Mountain View California USA","acronym":"MM '17"},"container-title":["Proceedings of the on Thematic Workshops of ACM Multimedia 2017"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3126686.3126720","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3126686.3126720","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T02:10:53Z","timestamp":1750212653000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3126686.3126720"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017,10,23]]},"references-count":33,"alternative-id":["10.1145\/3126686.3126720","10.1145\/3126686"],"URL":"https:\/\/doi.org\/10.1145\/3126686.3126720","relation":{},"subject":[],"published":{"date-parts":[[2017,10,23]]},"assertion":[{"value":"2017-10-23","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}