{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,7]],"date-time":"2026-03-07T14:01:46Z","timestamp":1772892106335,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":48,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,10,17]],"date-time":"2021-10-17T00:00:00Z","timestamp":1634428800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,10,17]]},"DOI":"10.1145\/3474085.3475465","type":"proceedings-article","created":{"date-parts":[[2021,10,18]],"date-time":"2021-10-18T20:00:05Z","timestamp":1634587205000},"page":"3192-3201","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":34,"title":["Cross-modal Retrieval and Synthesis (X-MRS): Closing the Modality Gap in Shared Subspace Learning"],"prefix":"10.1145","author":[{"given":"Ricardo","family":"Guerrero","sequence":"first","affiliation":[{"name":"Samsung AI Center, Cambridge, United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hai X.","family":"Pham","sequence":"additional","affiliation":[{"name":"Samsung AI Center, Cambridge, United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Vladimir","family":"Pavlovic","sequence":"additional","affiliation":[{"name":"Samsung AI Center, Cambridge, United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,10,17]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Quantifying Attention Flow in Transformers. arxiv","author":"Abnar Samira","year":"2005","unstructured":"Samira Abnar and Willem Zuidema . 2020. Quantifying Attention Flow in Transformers. arxiv : 2005 .00928 [cs.LG] Samira Abnar and Willem Zuidema. 2020. Quantifying Attention Flow in Transformers. arxiv: 2005.00928 [cs.LG]"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3209978.3210036"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240508.3240627"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1179"},{"key":"e_1_3_2_2_5_1","volume-title":"Kang Min Yoo, and Sang-goo Lee","author":"Choi Jihun","year":"2017","unstructured":"Jihun Choi , Kang Min Yoo, and Sang-goo Lee . 2017 . Learning to Compose Task-Specific Tree Structures. In In Association for the Advancement of Artificial Intelligence. www.aaai.org Jihun Choi, Kang Min Yoo, and Sang-goo Lee. 2017. Learning to Compose Task-Specific Tree Structures. In In Association for the Advancement of Artificial Intelligence. www.aaai.org"},{"key":"e_1_3_2_2_6_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. CoRR","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin , Ming-Wei Chang , Kenton Lee , and Kristina Toutanova . 2018 . BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. CoRR , Vol. abs\/ 1810 .04805 (2018). arxiv: 1810.04805 http:\/\/arxiv.org\/abs\/1810.04805 Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. CoRR, Vol. abs\/1810.04805 (2018). arxiv: 1810.04805 http:\/\/arxiv.org\/abs\/1810.04805"},{"key":"e_1_3_2_2_7_1","volume-title":"Dividing and Conquering Cross-Modal Recipe Retrieval: from Nearest Neighbours Baselines to SoTA. CoRR","author":"Fain Mikhail","year":"2019","unstructured":"Mikhail Fain , Andrey Ponikar , Ryan Fox , and Danushka Bollegala . 2019. Dividing and Conquering Cross-Modal Recipe Retrieval: from Nearest Neighbours Baselines to SoTA. CoRR , Vol. abs\/ 1911 .12763 ( 2019 ). arxiv: 1911.12763 http:\/\/arxiv.org\/abs\/1911.12763 Mikhail Fain, Andrey Ponikar, Ryan Fox, and Danushka Bollegala. 2019. Dividing and Conquering Cross-Modal Recipe Retrieval: from Nearest Neighbours Baselines to SoTA. CoRR, Vol. abs\/1911.12763 (2019). arxiv: 1911.12763 http:\/\/arxiv.org\/abs\/1911.12763"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01458"},{"key":"e_1_3_2_2_9_1","volume-title":"Imagine and Match: Improving Textual-Visual Cross-Modal Retrieval with Generative Models. In 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 7181--7189","author":"Gu J.","unstructured":"J. Gu , J. Cai , S. Joty , L. Niu , and G. Wang . 2018. Look , Imagine and Match: Improving Textual-Visual Cross-Modal Retrieval with Generative Models. In 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 7181--7189 . J. Gu, J. Cai, S. Joty, L. Niu, and G. Wang. 2018. Look, Imagine and Match: Improving Textual-Visual Cross-Modal Retrieval with Generative Models. In 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 7181--7189."},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"crossref","unstructured":"F. Han R. Guerrero and V. Pavlovic. 2020. CookGAN: Meal Image Synthesis from Ingredients. In WACV.  F. Han R. Guerrero and V. Pavlovic. 2020. CookGAN: Meal Image Synthesis from Ingredients. In WACV.","DOI":"10.1109\/WACV45572.2020.9093463"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.procs.2015.05.114"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00587"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2016.2598339"},{"key":"e_1_3_2_2_16_1","volume-title":"Adam: A Method for Stochastic Optimization. In International Conference on Learning Representation (ICLR).","author":"Diederik","unstructured":"Diederik P. Kingma and Jimmy Ba. 2015 . Adam: A Method for Stochastic Optimization. In International Conference on Learning Representation (ICLR). Diederik P. Kingma and Jimmy Ba. 2015. Adam: A Method for Stochastic Optimization. In International Conference on Learning Representation (ICLR)."},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.5555\/2969442.2969607"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01225-0_13"},{"key":"e_1_3_2_2_19_1","unstructured":"Tsung-Yi Lin Michael Maire Serge Belongie Lubomir Bourdev Ross Girshick James Hays Pietro Perona Deva Ramanan C. Lawrence Zitnick and Piotr Doll\u00e1r. 2015. Microsoft COCO: Common Objects in Context. arxiv: 1405.0312 [cs.CV]  Tsung-Yi Lin Michael Maire Serge Belongie Lubomir Bourdev Ross Girshick James Hays Pietro Perona Deva Ramanan C. Lawrence Zitnick and Piotr Doll\u00e1r. 2015. Microsoft COCO: Common Objects in Context. arxiv: 1405.0312 [cs.CV]"},{"key":"e_1_3_2_2_20_1","volume-title":"A Dataset for Learning Cross-Modal Embeddings for Cooking Recipes and Food Images","author":"Javier Mar\u00ed","year":"2019","unstructured":"Javier Mar\u00ed n, Aritro Biswas , Ferda Ofli , Nicholas Hynes , Amaia Salvador , Yusuf Aytar , Ingmar Weber , and Antonio Torralba . 2019. Recipe1M+ : A Dataset for Learning Cross-Modal Embeddings for Cooking Recipes and Food Images . IEEE Transactions on Pattern Analysis and Machine Intelligence ( 2019 ). Javier Mar\u00ed n, Aritro Biswas, Ferda Ofli, Nicholas Hynes, Amaia Salvador, Yusuf Aytar, Ingmar Weber, and Antonio Torralba. 2019. Recipe1M+: A Dataset for Learning Cross-Modal Embeddings for Cooking Recipes and Food Images. IEEE Transactions on Pattern Analysis and Machine Intelligence (2019)."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.5555\/2999792.2999959"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W19-8650"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.3043452"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-4009"},{"key":"e_1_3_2_2_25_1","volume-title":"CHEF: Cross-modal Hierarchical Embeddings for Food Domain Retrieval. In AAAI.","author":"Pham Hai Xuan","year":"2021","unstructured":"Hai Xuan Pham , Ricardo Guerrero , Jiatong Li , and Vladimir Pavlovic . 2021 . CHEF: Cross-modal Hierarchical Embeddings for Food Domain Retrieval. In AAAI. Hai Xuan Pham, Ricardo Guerrero, Jiatong Li, and Vladimir Pavlovic. 2021. CHEF: Cross-modal Hierarchical Embeddings for Food Domain Retrieval. In AAAI."},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.303"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.5555\/3045390.3045503"},{"key":"e_1_3_2_2_28_1","volume-title":"Xavier Giro i Nieto, and Adriana Romero","author":"Salvador Amaia","year":"2019","unstructured":"Amaia Salvador , Michal Drozdzal , Xavier Giro i Nieto, and Adriana Romero . 2019 . Inverse Cooking : Recipe Generation from Food Images. In CVPR. Amaia Salvador, Michal Drozdzal, Xavier Giro i Nieto, and Adriana Romero. 2019. Inverse Cooking: Recipe Generation from Food Images. In CVPR."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"crossref","unstructured":"Amaia Salvador Erhan Gundogdu Loris Bazzani and Michael Donoser. 2021. Revamping Cross-Modal Recipe Retrieval with Hierarchical Transformers and Self-supervised Learning. In CVPR.  Amaia Salvador Erhan Gundogdu Loris Bazzani and Michael Donoser. 2021. Revamping Cross-Modal Recipe Retrieval with Hierarchical Transformers and Self-supervised Learning. In CVPR.","DOI":"10.1109\/CVPR46437.2021.01522"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.327"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"crossref","unstructured":"R. Sennrich B. Haddow and A. Birch. 2016. Improving Neural Machine Translation Models with Monolingual Data. In ACL.  R. Sennrich B. Haddow and A. Birch. 2016. Improving Neural Machine Translation Models with Monolingual Data. In ACL.","DOI":"10.18653\/v1\/P16-1009"},{"key":"e_1_3_2_2_32_1","volume-title":"Manning","author":"Tai Kai Sheng","year":"2015","unstructured":"Kai Sheng Tai , Richard Socher , and Christopher D . Manning . 2015 . Improved Semantic Representations From Tree-Structured Long Short-Term Memory Networks. In Proceedings of the 53rd Annual Meeting of the Association for Computational Linguistics and the 7th International Joint Conference on Natural Language Processing (Volume 1: Long Papers). Association for Computational Linguistics, Beijing, China, 1556--1566. https:\/\/doi.org\/10.3115\/v1\/P15--1150 10.3115\/v1 Kai Sheng Tai, Richard Socher, and Christopher D. Manning. 2015. Improved Semantic Representations From Tree-Structured Long Short-Term Memory Networks. In Proceedings of the 53rd Annual Meeting of the Association for Computational Linguistics and the 7th International Joint Conference on Natural Language Processing (Volume 1: Long Papers). Association for Computational Linguistics, Beijing, China, 1556--1566. https:\/\/doi.org\/10.3115\/v1\/P15--1150"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/2380718.2380757"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.5555\/3295222.3295349"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"crossref","unstructured":"Hao Wang Guosheng Lin Steven C. H. Hoi and Chunyan Miao. 2020 a. Structure-Aware Generation Network for Recipe Generation from Images. In ECCV.  Hao Wang Guosheng Lin Steven C. H. Hoi and Chunyan Miao. 2020 a. Structure-Aware Generation Network for Recipe Generation from Images. In ECCV.","DOI":"10.1007\/978-3-030-58583-9_22"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01184"},{"key":"e_1_3_2_2_38_1","volume-title":"Ee peng Lim, and Steven C. H. Hoi. 2020 b. Cross-Modal Food Retrieval: Learning a Joint Embedding of Food Images and Recipes with Semantic Consistency and Attention Mechanism. arxiv","author":"Wang Hao","year":"2003","unstructured":"Hao Wang , Doyen Sahoo , Chenghao Liu , Ke Shu , Palakorn Achananuparp , Ee peng Lim, and Steven C. H. Hoi. 2020 b. Cross-Modal Food Retrieval: Learning a Joint Embedding of Food Images and Recipes with Semantic Consistency and Attention Mechanism. arxiv : 2003 .03955 [cs.CV] Hao Wang, Doyen Sahoo, Chenghao Liu, Ke Shu, Palakorn Achananuparp, Ee peng Lim, and Steven C. H. Hoi. 2020 b. Cross-Modal Food Retrieval: Learning a Joint Embedding of Food Images and Recipes with Semantic Consistency and Attention Mechanism. arxiv: 2003.03955 [cs.CV]"},{"key":"e_1_3_2_2_39_1","volume-title":"Morgan Funtowicz, and Jamie Brew.","author":"Wolf Thomas","year":"2019","unstructured":"Thomas Wolf , Lysandre Debut , Victor Sanh , Julien Chaumond , Clement Delangue , Anthony Moi , Pierric Cistac , Tim Rault , R\u00e9 mi Louf , Morgan Funtowicz, and Jamie Brew. 2019 . HuggingFace's Transformers: State-of-the-art Natural Language Processing. CoRR , Vol. abs\/ 1910 .03771 (2019). arxiv: 1910.03771 http:\/\/arxiv.org\/abs\/1910.03771 Thomas Wolf, Lysandre Debut, Victor Sanh, Julien Chaumond, Clement Delangue, Anthony Moi, Pierric Cistac, Tim Rault, R\u00e9 mi Louf, Morgan Funtowicz, and Jamie Brew. 2019. HuggingFace's Transformers: State-of-the-art Natural Language Processing. CoRR, Vol. abs\/1910.03771 (2019). arxiv: 1910.03771 http:\/\/arxiv.org\/abs\/1910.03771"},{"key":"e_1_3_2_2_40_1","volume-title":"Google's Neural Machine Translation System: Bridging the Gap between Human and Machine Translation. CoRR","author":"Wu Yonghui","year":"2016","unstructured":"Yonghui Wu , Mike Schuster , Zhifeng Chen , Quoc V. Le , Mohammad Norouzi , Wolfgang Macherey , Maxim Krikun , Yuan Cao , Qin Gao , Klaus Macherey , Jeff Klingner , Apurva Shah , Melvin Johnson , Xiaobing Liu , Lukasz Kaiser , Stephan Gouws , Yoshikiyo Kato , Taku Kudo , Hideto Kazawa , Keith Stevens , George Kurian , Nishant Patil , Wei Wang , Cliff Young , Jason Smith , Jason Riesa , Alex Rudnick , Oriol Vinyals , Greg Corrado , Macduff Hughes , and Jeffrey Dean . 2016. Google's Neural Machine Translation System: Bridging the Gap between Human and Machine Translation. CoRR , Vol. abs\/ 1609 .08144 ( 2016 ). arxiv: 1609.08144 http:\/\/arxiv.org\/abs\/1609.08144 Yonghui Wu, Mike Schuster, Zhifeng Chen, Quoc V. Le, Mohammad Norouzi, Wolfgang Macherey, Maxim Krikun, Yuan Cao, Qin Gao, Klaus Macherey, Jeff Klingner, Apurva Shah, Melvin Johnson, Xiaobing Liu, Lukasz Kaiser, Stephan Gouws, Yoshikiyo Kato, Taku Kudo, Hideto Kazawa, Keith Stevens, George Kurian, Nishant Patil, Wei Wang, Cliff Young, Jason Smith, Jason Riesa, Alex Rudnick, Oriol Vinyals, Greg Corrado, Macduff Hughes, and Jeffrey Dean. 2016. Google's Neural Machine Translation System: Bridging the Gap between Human and Machine Translation. CoRR, Vol. abs\/1609.08144 (2016). arxiv: 1609.08144 http:\/\/arxiv.org\/abs\/1609.08144"},{"key":"e_1_3_2_2_41_1","unstructured":"Q. Xie Z. Dai E. Hovy M. Luong and Q. Le. 2020. Unsupervised Data Augmentation for Consistency Training. In NeurIPS.  Q. Xie Z. Dai E. Hovy M. Luong and Q. Le. 2020. Unsupervised Data Augmentation for Consistency Training. In NeurIPS."},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00143"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.629"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.629"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"crossref","unstructured":"B. Zhu and Chong-Wah Ngo. 2020. CookGAN: Causality based Text-to-Image Synthesis. In CVPR.  B. Zhu and Chong-Wah Ngo. 2020. CookGAN: Causality based Text-to-Image Synthesis. In CVPR.","DOI":"10.1109\/CVPR42600.2020.00556"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01174"},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00595"},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.5555\/3045118.3045289"}],"event":{"name":"MM '21: ACM Multimedia Conference","location":"Virtual Event China","acronym":"MM '21","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 29th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3475465","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3474085.3475465","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:48:33Z","timestamp":1750193313000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3475465"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,10,17]]},"references-count":48,"alternative-id":["10.1145\/3474085.3475465","10.1145\/3474085"],"URL":"https:\/\/doi.org\/10.1145\/3474085.3475465","relation":{},"subject":[],"published":{"date-parts":[[2021,10,17]]},"assertion":[{"value":"2021-10-17","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}