{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T20:06:55Z","timestamp":1765310815084,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":41,"publisher":"ACM","funder":[{"name":"Ministry of Education, Singapore","award":["T2EP20222-0046"],"award-info":[{"award-number":["T2EP20222-0046"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755583","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:30:51Z","timestamp":1761377451000},"page":"6223-6231","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Mitigating Cross-modal Representation Bias for Multicultural Image-to-Recipe Retrieval"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-8878-8652","authenticated-orcid":false,"given":"Qing","family":"Wang","sequence":"first","affiliation":[{"name":"Singapore Management University, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4182-8261","authenticated-orcid":false,"given":"Chong-Wah","family":"Ngo","sequence":"additional","affiliation":[{"name":"Singapore Management University, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-2105-5419","authenticated-orcid":false,"given":"Yu","family":"Cao","sequence":"additional","affiliation":[{"name":"Singapore Management University, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0065-8665","authenticated-orcid":false,"given":"Ee-Peng","family":"Lim","sequence":"additional","affiliation":[{"name":"Singapore Management University, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Proceedings of the thirteenth language resources and evaluation conference. 6848--6854","author":"Carlsson Fredrik","year":"2022","unstructured":"Fredrik Carlsson, Philipp Eisen, Faton Rekathati, and Magnus Sahlgren. 2022. Cross-lingual and multilingual clip. In Proceedings of the thirteenth language resources and evaluation conference. 6848--6854."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_2_1","DOI":"10.1145\/3209978.3210036"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_3_1","DOI":"10.1007\/978-3-319-51811-4_48"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_4_1","DOI":"10.1145\/3240508.3240627"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_5_1","DOI":"10.1109\/WACV57701.2024.00800"},{"unstructured":"Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly et al. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020).","key":"e_1_3_2_1_6_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_7_1","DOI":"10.1109\/CVPR42600.2020.01458"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_8_1","DOI":"10.1145\/3474085.3475465"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_9_1","DOI":"10.1109\/CVPR.2016.90"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_10_1","DOI":"10.1145\/3581783.3612193"},{"doi-asserted-by":"publisher","unstructured":"Gabriel Ilharco Mitchell Wortsman Ross Wightman Cade Gordon Nicholas Carlini Rohan Taori Achal Dave Vaishaal Shankar Hongseok Namkoong John Miller Hannaneh Hajishirzi Ali Farhadi and Ludwig Schmidt. 2021. OpenCLIP. doi:10.5281\/zenodo.5143773","key":"e_1_3_2_1_11_1","DOI":"10.5281\/zenodo.5143773"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_12_1","DOI":"10.1109\/ICCV51070.2023.00371"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_13_1","DOI":"10.1109\/CVPR52688.2022.01751"},{"unstructured":"Shilong Liu Lei Zhang Xiao Yang Hang Su and Jun Zhu. 2021. Query2Label: A Simple Transformer Way to Multi-Label Classification. arXiv:2107.10834 [cs.CV]","key":"e_1_3_2_1_14_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_15_1","DOI":"10.1109\/CVPR52688.2022.00788"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_16_1","DOI":"10.1109\/CVPR52688.2022.01752"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_17_1","DOI":"10.1145\/3329168"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_18_1","DOI":"10.1109\/CVPR52688.2022.01606"},{"unstructured":"Judea Pearl. 2009. Causality. Cambridge university press.","key":"e_1_3_2_1_19_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_20_1","DOI":"10.1109\/CVPR42600.2020.01087"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_21_1","DOI":"10.1109\/ICCV48922.2021.00015"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_22_1","DOI":"10.1109\/CVPR.2019.01070"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_23_1","DOI":"10.1109\/CVPR46437.2021.01522"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_24_1","DOI":"10.1109\/CVPR.2017.327"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_25_1","DOI":"10.1162\/neco.1997.9.8.1735"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_26_1","DOI":"10.1109\/CVPRW56347.2022.00503"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_27_1","DOI":"10.1016\/j.cviu.2024.104071"},{"key":"e_1_3_2_1_28_1","volume-title":"Enhancing Recipe Retrieval with Foundation Models:ADataAugmentation Perspective. In European Conference on Computer Vision.","author":"Song Fangzhou","year":"2024","unstructured":"Fangzhou Song, Bin Zhu, Yanbin Hao, and Shuo Wang. 2024. Enhancing Recipe Retrieval with Foundation Models:ADataAugmentation Perspective. In European Conference on Computer Vision."},{"key":"e_1_3_2_1_29_1","volume-title":"NLLB-CLIP--train performant multilingual image retrieval model on a budget. arXiv preprint arXiv:2309.01859","author":"Visheratin Alexander","year":"2023","unstructured":"Alexander Visheratin. 2023. NLLB-CLIP--train performant multilingual image retrieval model on a budget. arXiv preprint arXiv:2309.01859 (2023)."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_30_1","DOI":"10.1109\/WACV57701.2024.00549"},{"key":"e_1_3_2_1_31_1","first-page":"3363","article-title":"Learning structural representations for recipe generation and food retrieval","volume":"45","author":"Wang Hao","year":"2022","unstructured":"Hao Wang, Guosheng Lin, Steven CH Hoi, and Chunyan Miao. 2022. Learning structural representations for recipe generation and food retrieval. IEEE Transactions on Pattern Analysis and Machine Intelligence 45, 3 (2022), 3363--3377.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_32_1","DOI":"10.1109\/CVPR.2019.01184"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_33_1","DOI":"10.1145\/3447548.3467249"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_34_1","DOI":"10.1145\/3485447.3512251"},{"key":"e_1_3_2_1_35_1","first-page":"1","article-title":"Learning text-image joint embedding for efficient cross-modal retrieval with deep feature engineering","volume":"40","author":"Xie Zhongwei","year":"2021","unstructured":"Zhongwei Xie, Ling Liu, Yanzhao Wu, Luo Zhong, and Lin Li. 2021. Learning text-image joint embedding for efficient cross-modal retrieval with deep feature engineering. ACM Transactions on Information Systems (TOIS) 40, 4 (2021), 1--27.","journal-title":"ACM Transactions on Information Systems (TOIS)"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_36_1","DOI":"10.1145\/3404835.3462823"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_37_1","DOI":"10.1038\/s41598-025-89461-8"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_38_1","DOI":"10.1109\/CVPR42600.2020.00556"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_39_1","DOI":"10.1145\/3512527.3531375"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_40_1","DOI":"10.1109\/CVPR.2019.01174"},{"key":"e_1_3_2_1_41_1","volume-title":"CREAMY: Cross-Modal Recipe Retrieval By Avoiding Matching Imperfectly","author":"Zou Zhuoyang","year":"2024","unstructured":"Zhuoyang Zou, Xinghui Zhu, Qinying Zhu, Yi Liu, and Lei Zhu. 2024. CREAMY: Cross-Modal Recipe Retrieval By Avoiding Matching Imperfectly. IEEE Access (2024)."}],"event":{"sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"acronym":"MM '25","name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755583","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T20:03:31Z","timestamp":1765310611000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755583"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":41,"alternative-id":["10.1145\/3746027.3755583","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755583","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}