{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T18:07:57Z","timestamp":1784138877920,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":21,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,20]]},"DOI":"10.1145\/3805712.3808431","type":"proceedings-article","created":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:06:26Z","timestamp":1784135186000},"page":"5108-5112","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["A General Framework for Multimodal LLM-Based Multimedia Understanding in Large-Scale Recommendation Systems"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-8627-8778","authenticated-orcid":false,"given":"Yiming","family":"Zhu","sequence":"first","affiliation":[{"name":"Meta Platforms, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-9531-5768","authenticated-orcid":false,"given":"Xu","family":"Liu","sequence":"additional","affiliation":[{"name":"Meta Platforms, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-9184-135X","authenticated-orcid":false,"given":"Ziyun","family":"Xu","sequence":"additional","affiliation":[{"name":"Meta Platforms, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-5419-3325","authenticated-orcid":false,"given":"Zheng","family":"Wu","sequence":"additional","affiliation":[{"name":"Meta Platforms, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5182-1234","authenticated-orcid":false,"given":"Joena","family":"Zhang","sequence":"additional","affiliation":[{"name":"Meta Platforms, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2756-5701","authenticated-orcid":false,"given":"Sirius","family":"Chen","sequence":"additional","affiliation":[{"name":"Meta Platforms, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-7151-965X","authenticated-orcid":false,"given":"Chenheli","family":"Hua","sequence":"additional","affiliation":[{"name":"Meta Platforms, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-1082-9055","authenticated-orcid":false,"given":"Silvester","family":"Yao","sequence":"additional","affiliation":[{"name":"Meta Platforms, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2676-1460","authenticated-orcid":false,"given":"Qichao","family":"Que","sequence":"additional","affiliation":[{"name":"Meta Platforms, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2616-6880","authenticated-orcid":false,"given":"Wentao","family":"Shi","sequence":"additional","affiliation":[{"name":"Meta Platforms, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-3253-5157","authenticated-orcid":false,"given":"Junfeng","family":"Pan","sequence":"additional","affiliation":[{"name":"Meta Platforms, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8350-392X","authenticated-orcid":false,"given":"Linhong","family":"Zhu","sequence":"additional","affiliation":[{"name":"Meta Platforms, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al.","author":"Achiam Josh","year":"2023","unstructured":"Josh Achiam, Steven Adler, Sandhini Agarwal, Lama Ahmad, Ilge Akkaya, Florencia Leoni Aleman, Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al., 2023. Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_2_1","first-page":"23716","article-title":"Flamingo: a visual language model for few-shot learning","volume":"35","author":"Alayrac Jean-Baptiste","year":"2022","unstructured":"Jean-Baptiste Alayrac, Jeff Donahue, Pauline Luc, Antoine Miech, Iain Barr, Yana Hasson, Karel Lenc, Arthur Mensch, Katherine Millican, Malcolm Reynolds, et al., 2022. Flamingo: a visual language model for few-shot learning. NeurIPS, Vol. 35 (2022), 23716-23736.","journal-title":"NeurIPS"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3716393"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1155\/2018\/5497070"},{"key":"e_1_3_2_1_5_1","volume-title":"Llama 2: Open Foundation and Fine-Tuned Chat Models. arXiv preprint arXiv:2307.09288","author":"Meta","year":"2023","unstructured":"Meta GenAI. 2023. Llama 2: Open Foundation and Fine-Tuned Chat Models. arXiv preprint arXiv:2307.09288 (2023)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3485447.3512079"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.9973"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/2648584.2648589"},{"key":"e_1_3_2_1_9_1","volume-title":"International conference on machine learning. PMLR, 4904-4916","author":"Jia Chao","year":"2021","unstructured":"Chao Jia, Yinfei Yang, Ye Xia, Yi-Ting Chen, Zarana Parekh, Hieu Pham, Quoc Le, Yun-Hsuan Sung, Zhen Li, and Tom Duerig. 2021. Scaling up visual and vision-language representation learning with noisy text supervision. In International conference on machine learning. PMLR, 4904-4916."},{"key":"e_1_3_2_1_10_1","volume-title":"International conference on machine learning. PMLR","author":"Li Junnan","year":"2023","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven Hoi. 2023. Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In International conference on machine learning. PMLR, 19730-19742."},{"key":"e_1_3_2_1_11_1","unstructured":"Tomas Mikolov Kai Chen Greg Corrado and Jeffrey Dean. 2013. Efficient Estimation of Word Representations in Vector Space. arXiv:1301.3781 [cs.CL] https:\/\/arxiv.org\/abs\/1301.3781"},{"key":"e_1_3_2_1_12_1","unstructured":"Maxim Naumov Dheevatsa Mudigere Hao-Jun Michael Shi Jianyu Huang Narayanan Sundaraman Jongsoo Park Xiaodong Wang Udit Gupta Carole-Jean Wu Alisson G. Azzolini Dmytro Dzhulgakov Andrey Mallevich Ilia Cherniavskii Yinghai Lu Raghuraman Krishnamoorthi Ansha Yu Volodymyr Kondratenko Stephanie Pereira Xianjie Chen Wenlin Chen Vijay Rao Bill Jia Liang Xiong and Misha Smelyanskiy. 2019. Deep Learning Recommendation Model for Personalization and Recommendation Systems. arXiv:1906.00091 [cs.IR] https:\/\/arxiv.org\/abs\/1906.00091"},{"key":"e_1_3_2_1_13_1","volume-title":"International conference on machine learning. PmLR, 8748-8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al., 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PmLR, 8748-8763."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645458"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.3390\/fi17060252"},{"key":"e_1_3_2_1_16_1","volume-title":"Language Models Encode Collaborative Signals in Recommendation. CoRR","author":"Sheng Leheng","year":"2024","unstructured":"Leheng Sheng, An Zhang, Yi Zhang, Yuxin Chen, Xiang Wang, and Tat-Seng Chua. 2024. Language Models Encode Collaborative Signals in Recommendation. CoRR (2024)."},{"key":"e_1_3_2_1_17_1","volume-title":"Jonas Meinertz Hansen, and Ole Winther","author":"Svenstrup Dan","year":"2017","unstructured":"Dan Svenstrup, Jonas Meinertz Hansen, and Ole Winther. 2017. Hash Embeddings for Efficient Word Representations. arXiv:1709.03933 [cs.CL] https:\/\/arxiv.org\/abs\/1709.03933"},{"key":"e_1_3_2_1_18_1","unstructured":"Gemini Team Rohan Anil Sebastian Borgeaud Jean-Baptiste Alayrac Jiahui Yu Radu Soricut Johan Schalkwyk Andrew M Dai Anja Hauth Katie Millican et al. 2023. Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805 (2023)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543507.3583206"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3708882"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3611380.3628561"}],"event":{"name":"SIGIR '26: The 49th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:30:41Z","timestamp":1784136641000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805712.3808431"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":21,"alternative-id":["10.1145\/3805712.3808431","10.1145\/3805712"],"URL":"https:\/\/doi.org\/10.1145\/3805712.3808431","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}